diff --git a/ddprof-lib/src/main/cpp/counters.h b/ddprof-lib/src/main/cpp/counters.h index 3293d07737..1f9cdbcd67 100644 --- a/ddprof-lib/src/main/cpp/counters.h +++ b/ddprof-lib/src/main/cpp/counters.h @@ -183,6 +183,26 @@ * and re-emits it on a later dump while the leak candidate is still \ * live. */ \ X(REFERENCE_CHAIN_WRITE_DROPPED, "reference_chain_write_dropped") \ + /* LivenessTracker::releaseLeakTag() was called for a leak-tag slot that \ + * is already free (zero/zero encoding) - a double release. The release is \ + * dropped: pushing the index twice would let acquireLeakTag() hand the \ + * same tag to two live objects. A nonzero rate here means the leak-tag \ + * ownership accounting (tagLeakInstances()' tag-adoption branch) is \ + * sharing one pool tag between entries. */ \ + X(REFERENCE_CHAIN_LEAK_TAG_DOUBLE_RELEASE, "reference_chain_leak_tag_double_release") \ + /* LivenessTracker urgency-boost observability (admitForTracking()): \ + * ADMITS counts every 100% admission made while _urgent_tracking is set \ + * and the table is below its cap; BACKED_OFF counts admissions at the \ + * high-water mark, where the boost degrades to watched-tid-only and the \ + * thread falls back to the configured subsample ratio. A persistently \ + * rising BACKED_OFF means the urgency window is outpacing the cleanup \ + * reaper. */ \ + X(LIVENESS_URGENT_BOOST_ADMITS, "liveness_urgent_boost_admits") \ + X(LIVENESS_URGENT_BOOST_BACKED_OFF, "liveness_urgent_boost_backed_off") \ + /* Defensive cap: releaseLeakTag() found the free list full. Unreachable \ + * by construction while every release is paired with an acquire; a \ + * nonzero value means the acquire/release pairing is broken somewhere. */ \ + X(REFERENCE_CHAIN_LEAK_TAG_RELEASE_OVERFLOW, "reference_chain_leak_tag_release_overflow") \ /* FrontierTable's own calloc/realloc-backed storage (referenceChains.cpp) - \ * outside NMT's visibility since it bypasses os::malloc, so this is the only \ * way to attribute its native RSS contribution. */ \ diff --git a/ddprof-lib/src/main/cpp/javaApi.cpp b/ddprof-lib/src/main/cpp/javaApi.cpp index 90f6696755..77c04e6bfd 100644 --- a/ddprof-lib/src/main/cpp/javaApi.cpp +++ b/ddprof-lib/src/main/cpp/javaApi.cpp @@ -1104,6 +1104,223 @@ Java_com_datadoghq_profiler_JavaProfiler_dumpContext(JNIEnv* env, jclass unused) TEST_LOG("===> Context: tid:%lu, spanId=%lu, rootSpanId=%lu", OS::threadId(), spanId, rootSpanId); } +// LivenessTracker/ReferenceChainTracker test seams. Unlike +// testlog()/dumpContext() above (harmless no-ops in release, via TEST_LOG's +// own release-mode expansion to nothing), these mutate real tracker state +// (tagging objects, seeding population history) - shipping them into a +// release build would let a caller corrupt the actual leak-detection state, +// not just add a silent no-op. Guarded out entirely instead, so they only +// exist in the debug build ddprof-test's `testdebug` Gradle task loads +// (`-DDEBUG`, see ConfigurationPresets.kt's configureDebug()) - never in the +// `-DNDEBUG` release build. +#ifdef DEBUG +#include "livenessTracker.h" +#include "referenceChains.h" +#include + +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setGcGenerationsEnabled0( + JNIEnv *env, jclass unused, jboolean enabled) { + LivenessTracker::instance()->setGcGenerationsForTest(enabled); + return JNI_TRUE; +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_seedKlassPopulationSample0( + JNIEnv *env, jclass unused, jint klassId, jint count, jlong epoch) { + int slot; + bool created; + LivenessTracker::instance()->klassPopulationRecordForTest( + (u32)klassId, (u16)count, (u64)epoch, &slot, &created); +} + +// Seeds one per-(klass, tid) trend sample - see tidTrendRecordForTest()'s +// own comment (livenessTracker.h) for the synthetic-flag exemption and the +// real-tid requirement scenarios must honor. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_seedTidTrendSample0( + JNIEnv *env, jclass unused, jint klassId, jint tid, jint count, + jlong epoch) { + LivenessTracker::instance()->tidTrendRecordForTest( + (u32)klassId, (jint)tid, (u32)count, (u64)epoch); +} + +// Wires a real, caller-chosen live object in as klassId's leak-candidate +// representative, so a test-seeded slope signal (seedKlassPopulationSample0 +// above) and a directly-tagged frontier root (tagAsReferenceChainRoot0 +// below) can be joined into one deterministic end-to-end run of +// pollWatchedTargets()'s bridging step - without either LivenessTracker's +// real allocation sampler or ReferenceChainTracker's root-seeded walk ever +// running. Takes its own weak global ref (klassPopulationSetRepresentativeForTest()'s +// own contract, livenessTracker.h) rather than aliasing any handle the +// caller manages. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setKlassPopulationRepresentativeForTest0( + JNIEnv *env, jclass unused, jint klassId, jobject representative) { + jweak rep = env->NewWeakGlobalRef(representative); + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( + env, (u32)klassId, rep); +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_resetKlassPopulationForTest0( + JNIEnv *env, jclass unused) { + LivenessTracker::instance()->klassPopulationResetForTest(); +} + +extern "C" DLLEXPORT jintArray JNICALL +Java_com_datadoghq_profiler_JavaProfiler_selectLeakCandidateKlassIds0( + JNIEnv *env, jclass unused) { + KlassCandidate candidates[5]; + int n = LivenessTracker::instance()->selectLeakCandidates(candidates, 5); + jintArray result = env->NewIntArray(n); + if (result == nullptr || n == 0) { + return result; + } + jint ids[5]; + for (int i = 0; i < n; i++) { + ids[i] = (jint)candidates[i].klass_id; + } + env->SetIntArrayRegion(result, 0, n, ids); + return result; +} + +extern "C" DLLEXPORT jlong JNICALL +Java_com_datadoghq_profiler_JavaProfiler_tagAsReferenceChainRoot0( + JNIEnv *env, jclass unused, jobject target) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr) { + return 0; + } + return ReferenceChainTracker::instance()->tagAsRootForTest(jvmti, env, + target); +} + +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfiler_runReferenceChainPass0( + JNIEnv *env, jclass unused) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr) { + return JNI_FALSE; + } + return ReferenceChainTracker::instance()->runPassSerialized(jvmti, env); +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_pollReferenceChainTargets0( + JNIEnv *env, jclass unused) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr) { + return; + } + ReferenceChainTracker::instance()->pollWatchedTargetsSerialized(jvmti, env); +} + +extern "C" DLLEXPORT jint JNICALL +Java_com_datadoghq_profiler_JavaProfiler_drainReferenceChainEventCount0( + JNIEnv *env, jclass unused) { + std::vector events; + ReferenceChainTracker::instance()->drainPendingChainEvents(&events); + return (jint)events.size(); +} + +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_resetReferenceChainSearchForTest0( + JNIEnv *env, jclass unused) { + jvmtiEnv *jvmti = VM::jvmti(); + ReferenceChainTracker::instance()->resetSearchStateForTest(jvmti, env); +} + +// Diagnostic-only: reads target's existing JVMTI tag (does NOT tag it - +// unlike tagAsReferenceChainRoot0 above, a target the real search has not +// reached yet must be left untagged) and reports its FIFO distance from the +// front of ReferenceChainTracker's pending-expansion queue. See +// ReferenceChainTracker::pendingExpandPositionForTest()'s own comment for +// the return-value contract. +extern "C" DLLEXPORT jlong JNICALL +Java_com_datadoghq_profiler_JavaProfiler_getReferenceChainPendingPositionForTest0( + JNIEnv *env, jclass unused, jobject target) { + jvmtiEnv *jvmti = VM::jvmti(); + if (jvmti == nullptr || target == nullptr) { + return -2; + } + jlong tag = 0; + jvmtiError err = jvmti->GetTag(target, &tag); + if (err != JVMTI_ERROR_NONE) { + return -2; + } + return (jlong)ReferenceChainTracker::instance()->pendingExpandPositionForTest( + tag); +} + +extern "C" DLLEXPORT jlong JNICALL +Java_com_datadoghq_profiler_JavaProfiler_getReferenceChainPendingSizeForTest0( + JNIEnv *env, jclass unused) { + return (jlong)ReferenceChainTracker::instance()->pendingExpandSizeForTest(); +} + +// Seeds one heap-floor-ring sample directly (LivenessTracker::secondsToOOM()'s +// input), bypassing the real GarbageCollectionFinish callback - lets a test +// build an arbitrary rising/flat heap-usage-over-time history without +// waiting on real GCs. timestampNs values are only ever compared against +// each other (secondsToOOM()'s own ringWindowStats() deltas), never against +// a real wall clock, so a test may use any self-consistent, strictly +// increasing sequence. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_heapFloorRecordForTest0( + JNIEnv *env, jclass unused, jlong usedBytes, jlong timestampNs) { + LivenessTracker::instance()->heapFloorRecordForTest((u64)usedBytes, + (u64)timestampNs); +} + +// Bypasses initialize_table()'s JNI-dependent HeapUsage::getMaxHeap() call so +// secondsToOOM() can be exercised against a test-chosen fake max heap size, +// independent of whatever -Xmx this JVM's own shared, no-forkEvery fork +// happens to run with. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setMaxHeapBytesForTest0( + JNIEnv *env, jclass unused, jlong maxHeapBytes) { + LivenessTracker::instance()->setMaxHeapBytesForTest((jlong)maxHeapBytes); +} + +// Temporarily disables onGC()'s own recordHeapFloorSample() call so a test +// can seed the heap-floor ring exclusively via heapFloorRecordForTest0() +// without a real GC interleaving a sample with a real OS::nanotime() +// timestamp and real heap usage, corrupting secondsToOOM()'s projection. +extern "C" DLLEXPORT void JNICALL +Java_com_datadoghq_profiler_JavaProfiler_setHeapFloorRecordingForTest0( + JNIEnv *env, jclass unused, jboolean enabled) { + LivenessTracker::instance()->setHeapFloorRecordingForTest(enabled == JNI_TRUE); +} + +// Exposes ReferenceChainTracker::shouldRunPass() directly (see that seam's +// own comment, referenceChains.h) - unlike runReferenceChainPass0() above, +// which calls runPass() unconditionally, this reports whether the +// search-restart gate itself (canAffordNewSearch() -> hasLeakSignal()) would +// currently allow a fresh/terminal search to start. +extern "C" DLLEXPORT jboolean JNICALL +Java_com_datadoghq_profiler_JavaProfiler_shouldRunPassForTest0(JNIEnv *env, + jclass unused) { + return ReferenceChainTracker::instance()->shouldRunPassForTest( + OS::nanotime()) + ? JNI_TRUE + : JNI_FALSE; +} + +// Exposes ReferenceChainTracker::passesRun() directly - not itself DEBUG-gated on the native side +// (used by production JFR event fields too), but exposed here only for test use: lets a test note +// the current pass count before creating an object, then wait for that count to advance before +// trusting any match against it - the only way to be certain a match came from a pass whose own +// expandFrontier() (and therefore collectStaleExpandedEntriesForRotation()) ran strictly after the +// object existed, rather than from the same pass racing the object's creation. +extern "C" DLLEXPORT jint JNICALL +Java_com_datadoghq_profiler_JavaProfiler_referenceChainPassesRunForTest0( + JNIEnv *env, jclass unused) { + return (jint)ReferenceChainTracker::instance()->passesRun(); +} + +#endif // DEBUG + // ---- Test-only reads of the current thread's OTEP record ----------------------------------- // Each reads the current carrier's record directly via ProfiledThread::current(), with no // detach/attach (diagnostic-only, not on any signal-handler or hot write path). diff --git a/ddprof-lib/src/main/cpp/livenessTracker.cpp b/ddprof-lib/src/main/cpp/livenessTracker.cpp index 5bfdb9aef8..51369cf09d 100644 --- a/ddprof-lib/src/main/cpp/livenessTracker.cpp +++ b/ddprof-lib/src/main/cpp/livenessTracker.cpp @@ -13,6 +13,7 @@ #include "common.h" #include "context.h" #include "context_api.h" +#include "counters.h" #include "hotspot/vmStructs.h" #include "hotspot/vmStructs.inline.h" #include "incbin.h" @@ -39,54 +40,35 @@ constexpr int LivenessTracker::MIN_SAMPLING_INTERVAL; namespace { -// Max surviving entries whose per-epoch JNI class resolution (resolveKlassId: -// GetObjectClass + Class.getName() + StringDictionary lookup) a single -// cleanup_table() sweep performs under the exclusive _table_lock. Entries -// beyond the budget fall back to their cached class id (the exact semantics -// the allow_resolve=false path already uses); the next epoch's sweep resolves -// the next tranche, and cached ids accumulated across sweeps keep most -// entries resolved anyway. Bounds the sweep's exclusive-lock window so it -// stays proportional to table bookkeeping rather than to survivor count - -// every shared-lock scanner (tagLeakInstances(), getLiveTraceIds()) is -// blocked for the whole sweep otherwise. -constexpr u32 RESOLVE_BUDGET_PER_SWEEP = 256; - -// Trend statistics of a chronological ring window - the one computation +// Window aggregation for a chronological ring - the one computation // hasQualifyingGrowth() (per-klass count_ring) and heapFloorRising() (the // aggregate _heap_floor_ring) both need, factored out so the window/index -// derivation and the two aggregation loops exist in exactly one place rather -// than three near-identical copies. Templated on the reader rather than the -// ring's element type or storage: the per-klass ring is a plain array read -// under the caller's already-held _table_lock, while the heap-floor ring is -// lock-free and read via loadAcquire() (see _heap_floor_ring's own comment, -// livenessTracker.h) - `read(i)` lets each caller supply its own access -// discipline for physical slot `i` without this shared loop needing to know -// which one applies. -// -// DESPITE THE NAME, this is not thirds statistics: the design doc's original -// "mean of earliest third vs mean of recent third" comparison was replaced by -// a full-window least-squares linear regression (see ringThirdsStats below), -// which uses all samples and is far more robust for oscillating-but-growing -// trends. The field names survive as the regression values consumers treat -// as the window's "earliest"/"recent" levels: -// earliest_mean - regression value at the window's OLDEST sample (x = 0); -// recent_mean - regression value at the window's NEWEST sample (x = fill-1); -// earliest_min - true minimum over the FULL window; -// recent_min - true minimum over the most recent HALF of the window. -// Consumers read earliest_mean/recent_mean as a smoothed start-vs-end delta -// (a regression slope over the window's span) and earliest_min/recent_min as -// floor checks. Renaming the fields would touch every consumer for no -// behavioral change, so the mapping is documented here instead. -struct RingThirdsStats { - double earliest_mean; // regression value at the window's oldest sample - double recent_mean; // regression value at the window's newest sample - double earliest_min; // true min over the full window - double recent_min; // true min over the window's most recent half +// derivation and the aggregation loops exist in exactly one place rather +// than three near-identical copies. The name is historical: despite the +// struct's field names, this computes a FULL-WINDOW least-squares linear +// regression, not third means - see ringWindowStats()'s comment below. Templated on +// the reader rather than the ring's element type or storage: the per-klass +// ring is a plain array read under the caller's already-held _table_lock, +// while the heap-floor ring is lock-free and read via loadAcquire() (see +// _heap_floor_ring's own comment, livenessTracker.h) - `read(i)` lets each +// caller supply its own access discipline for physical slot `i` without +// this shared loop needing to know which one applies. +struct RingWindowStats { + // Fitted-line values, NOT window means: earliest_mean is the regression + // intercept (the fitted value at x=0, the oldest sample), recent_mean the + // fitted endpoint (x=n-1, the newest). earliest_min/recent_min are true + // minima over the first/second half of the window (recent_min's half + // boundary is i >= n/2, so for odd n the median sample joins the recent + // half). + double earliest_mean; + double recent_mean; + double earliest_min; + double recent_min; }; template -bool ringThirdsStats(int head, int fill, int ring_size, int min_fill, - Reader read, RingThirdsStats *out) { +bool ringWindowStats(int head, int fill, int ring_size, int min_fill, + Reader read, RingWindowStats *out) { if (fill < min_fill) { return false; } @@ -133,21 +115,21 @@ bool ringThirdsStats(int head, int fill, int ring_size, int min_fill, // Recent-half corroboration for a usage ring (see secondsToOOM()'s own // comment): a rising full-window trend whose most recent half is flat is a -// plateaued step change, not ongoing growth. The recent half's own -// regression (see ringThirdsStats) must show a strictly positive delta for -// the full-window trend to stand. A too-sparse recent half (below min_fill) -// REJECTS the projection rather than letting it through: false means -// "reject". Deliberately stricter than the single-ring version's -// have_recent_half semantics (which only rejected on a confirmed flat -// recent half) - with the ring-fill floors the caller already enforces, a -// sparse recent half means the recent data does not yet support the trend, -// so the boundary projection waits for more samples. +// plateaued step change, not ongoing growth. Returns true only when the +// recent half itself shows a rising trend; an absent or too-sparse recent +// half REJECTS the boundary (returns false). That is the conservative +// direction: a boundary projection that only the full window supports can +// mask a dip-then-recover shape whose full-window endpoints happen to +// agree, and the OOM urgency ramp is expensive enough to demand +// corroboration before firing. (An earlier draft of this comment claimed +// the sparse case lets the full-window trend stand alone - the code never +// did that; the comment was wrong.) template bool corroborateRecentHalf(u8 head, u8 fill, int ring_size, int min_fill, Reader read) { int half_fill = fill / 2; - RingThirdsStats recent_half_stats; - bool have_half = ringThirdsStats(head, half_fill, ring_size, min_fill, read, + RingWindowStats recent_half_stats; + bool have_half = ringWindowStats(head, half_fill, ring_size, min_fill, read, &recent_half_stats); double half_delta = have_half ? recent_half_stats.recent_mean - recent_half_stats.earliest_mean @@ -157,7 +139,8 @@ bool corroborateRecentHalf(u8 head, u8 fill, int ring_size, int min_fill, } // namespace -void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { +void LivenessTracker::cleanup_table(bool forced, bool allow_resolve, + bool account_epoch) { u64 current = load(_last_gc_epoch); u64 target_gc_epoch = load(_gc_epoch); TEST_LOG_SUMMARY("LivenessTracker::cleanup_table forced=%d gc_generations=%d current_epoch=%llu " @@ -187,23 +170,27 @@ void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { _table_lock.lock(); u64 claimed = load(_last_gc_epoch); - // Forward-only claim: a cleanup caller captured target_gc_epoch BEFORE it - // took the lock; by the time it holds the lock another caller may already - // have published a NEWER epoch (target 6 published while this call's - // snapshot still says 5). The old `!=`-gated unconditional CAS would move - // _last_gc_epoch BACKWARD 6->5, making epoch_diff negative and wrapping - // every survivor's unsigned age below - folding epochs out of order and - // manufacturing leak candidates / duplicate population samples. Only the - // caller whose target is strictly newer claims; a stale snapshot skips - // epoch accounting entirely (its survivors are still swept, their ages - // simply do not move for this no-op epoch). - bool is_epoch_owner = target_gc_epoch > claimed && + // account_epoch=false (track()'s table-overflow branch) makes this a + // pure reaper: it must NOT claim the epoch (otherwise the background + // sweep's fold for this epoch would be suppressed by the claimed + // _last_gc_epoch while this call never folds it - the epoch's population + // sample would be lost), must not age survivors (that accounting belongs + // to the epoch-advance pass), and must not touch the klass-population + // scratch (no fold will consume it here). It only reaps collected + // entries, which is what the overflow path needs to free table slots. + bool is_epoch_owner = account_epoch && target_gc_epoch != claimed && __atomic_compare_exchange_n(&_last_gc_epoch, &claimed, target_gc_epoch, false, __ATOMIC_RELAXED, __ATOMIC_RELAXED); - // On a lost CAS race `claimed` holds the actual current epoch (>= our - // stale target), so the raw diff is <= 0 for every non-owner; clamp so a - // survivor's unsigned age can never wrap. int epoch_diff = (int)(target_gc_epoch - claimed); + // A forced sweep can lose the epoch race: between this call's + // target_gc_epoch snapshot and the lock acquisition, a GC callback on + // another thread claimed a NEWER epoch (claimed > target_gc_epoch), + // making this raw difference negative. Aging survivors by a negative + // diff would rewind their ages and corrupt the Lindy oldest[] ordering + // and the generation-count signal. The newer claim already advanced + // every age by the full inter-epoch step, so this sweep contributes no + // aging of its own - clamp to zero. (The epoch-ownership CAS above is + // unaffected: is_epoch_owner already correctly reports false here.) if (epoch_diff < 0) { epoch_diff = 0; } @@ -231,7 +218,7 @@ void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { } } _klass_population_size = 0; - _klass_count_scratch_size = 0; + klassCountScratchReset(); _last_class_map_generation = current_class_map_generation; } @@ -239,9 +226,6 @@ void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { if (sz > 0) { u64 start = OS::nanotime(), end; u32 newsz = 0; - // Per-sweep JNI-resolution budget - see the survivor loop's allow_resolve - // comment below for why the exclusive-lock window must stay bounded. - u32 resolve_budget = RESOLVE_BUDGET_PER_SWEEP; std::set kept_classes; for (u32 i = 0; i < sz; i++) { if (_table[i].ref != nullptr && @@ -263,8 +247,10 @@ void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { // session. Gated on is_epoch_owner (not !forced) so a forced // (table-overflow) sweep still contributes one population sample // per genuinely new GC epoch instead of silently dropping it. + // account_epoch=false sweeps never reach this (is_epoch_owner is + // forced false there). u32 klass_id = 0; - if (allow_resolve && resolve_budget > 0) { + if (allow_resolve && _table[target].cached_klass_id == 0) { // GetObjectClass + Class.getName() + StringDictionary lookup per // surviving entry, previously paid only at JFR-flush time (see // flush_table() below). Only affordable off the allocation-hot @@ -273,17 +259,16 @@ void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { // both pass allow_resolve=true; track()'s hot-path forced sweep // does not (see cleanup_table()'s own header comment). // - // Bounded per sweep (RESOLVE_BUDGET_PER_SWEEP): every resolution - // here runs under the EXCLUSIVE _table_lock, so an unbounded - // survivor count would stretch this sweep's critical section - // proportionally to the population (blocking every shared-lock - // scanner for the whole per-survivor JNI sequence). Entries past - // the budget fall through to the cached-id path below - the - // exact semantics the allow_resolve=false path already accepts - - // and the next epoch's sweep resolves the next tranche; the - // cached ids accumulated across sweeps keep most entries - // resolved anyway. - resolve_budget--; + // The cached_klass_id == 0 gate is what keeps this affordable + // while holding the EXCLUSIVE table lock: an entry's class is + // immutable, so once resolved its id never changes (until the + // class-map generation reset above zeroes the whole cache). + // Steady state costs one u32 read per survivor; the full + // NewLocalRef + resolveKlassId() JNI round-trip is paid once + // per entry per class-map generation. Before this gate the + // per-survivor round-trip ran on EVERY sweep for entries whose + // resolution failed, and for every entry whenever + // _gc_generations was re-enabled. jobject ref = env->NewLocalRef(_table[target].ref); if (ref != nullptr) { klass_id = resolveKlassId(env, ref); @@ -301,16 +286,13 @@ void LivenessTracker::cleanup_table(bool forced, bool allow_resolve) { env->DeleteLocalRef(ref); } } else { - // track()'s table-overflow branch calls cleanup_table(true, - // false) synchronously from the allocation-sampling call stack - // (JVMTI SampledObjectAlloc callback). resolveKlassId() calls - // Class.getName(), a genuine Java-bytecode upcall (unlike the - // plain native jvmti->GetClassSignature() call - // ObjectSampler::recordAllocation already makes on this same - // callback stack) - too costly, and too re-entrancy-prone via - // the String allocation it can trigger, to run from there. Reuse - // whatever class id an earlier resolving sweep already resolved - // for this entry instead; if it was never resolved, this entry's + // Either a non-resolving sweep (track()'s table-overflow branch + // calls cleanup_table(true, false) synchronously from the + // allocation-sampling call stack - JVMTI SampledObjectAlloc + // callback; resolveKlassId() calls Class.getName(), a genuine + // Java-bytecode upcall, too costly and re-entrancy-prone from + // there), or an already-resolved entry on a resolving sweep. + // Reuse the cached id; if it was never resolved, this entry's // sample for this epoch is dropped rather than resolving now. klass_id = _table[target].cached_klass_id; } @@ -426,14 +408,10 @@ void LivenessTracker::insertOldestSample(KlassCountScratch &scratch, } jlong LivenessTracker::acquireLeakTag(u64 call_trace_id, jint tid) { - // Pool mutation is serialized by its own lock, not by _table_lock: this is - // called under the SHARED table lock (tagLeakInstances, BFS poll thread) - // while releaseLeakTag() runs under the EXCLUSIVE table lock (cleanup_table, - // GC-callback thread) - shared vs exclusive excludes those two from each - // other, but any future second shared-lock mutator would corrupt the LIFO - // free list. The dedicated lock keeps the pool correct independent of which - // table lock mode the caller holds. Lock order: _table_lock (any mode) is - // always acquired BEFORE _leak_tag_pool_lock, never the reverse. + // Pool lock: see _leak_tag_pool_lock's comment. Callers hold various + // combinations of the table lock (exclusive in cleanup_table's reaper, + // none in tagLeakInstances' unlocked tagging phase), so the pool cannot + // rely on either. _leak_tag_pool_lock.lock(); if (_leak_tag_free_count <= 0) { _leak_tag_pool_lock.unlock(); @@ -451,9 +429,28 @@ void LivenessTracker::releaseLeakTag(jlong tag) { return; } int idx = (int)(tag - LEAK_TAG_BASE); - // See acquireLeakTag()'s comment for the dedicated pool lock (and the - // _table_lock -> _leak_tag_pool_lock ordering). _leak_tag_pool_lock.lock(); + // Double-release guard: a zero/zero slot is free (see getLeakTagInfo()'s + // encoding note). Releasing an already-free tag would push its index onto + // the free list twice; a later acquireLeakTag() would then hand the same + // tag to two live objects and corrupt leak attribution. Found reachable + // in review: tagLeakInstances()' tag-adoption branch could let two table + // entries share one pool tag, so the second release hit this path. + if (_leak_tag_info[idx].call_trace_id == 0 && _leak_tag_info[idx].tid == 0) { + _leak_tag_pool_lock.unlock(); + Counters::increment(REFERENCE_CHAIN_LEAK_TAG_DOUBLE_RELEASE); + return; + } + // Bounds guard: never push past the pool. With the double-release guard + // above this is unreachable by construction (each index is pushed at most + // once between rebuilds), but the rebuild path (rebuildLeakTagFreeList) + // rewrites the whole list anyway, so a defensive cap here is cheap and + // keeps a future accounting bug from overflowing the array. + if (_leak_tag_free_count >= LEAK_TAG_POOL_SIZE) { + _leak_tag_pool_lock.unlock(); + Counters::increment(REFERENCE_CHAIN_LEAK_TAG_RELEASE_OVERFLOW); + return; + } _leak_tag_info[idx].call_trace_id = 0; _leak_tag_info[idx].tid = 0; _leak_tag_free_list[_leak_tag_free_count++] = idx; @@ -461,22 +458,25 @@ void LivenessTracker::releaseLeakTag(jlong tag) { } bool LivenessTracker::getLeakTagInfo(jlong tag, u64 *out_call_trace_id, - jint *out_tid) const { + jint *out_tid) { if (tag < LEAK_TAG_BASE || tag >= LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE) { return false; } int idx = (int)(tag - LEAK_TAG_BASE); + // The two slot fields are written by acquireLeakTag() and releaseLeakTag() + // under the pool lock, from several distinct thread contexts + // (cleanup_table()'s reaper pass via flush_table()'s JFR cadence, + // track()'s overflow branch, maybeForceCleanup()'s background tick, and + // tagLeakInstances()' unlocked tagging phase). Read them under the same + // lock so a torn acquire/release cannot be observed as a spurious + // zero/zero (which would read as "not in use"). + _leak_tag_pool_lock.lock(); // releaseLeakTag() zeroes both fields, so a zero/zero slot means the tag // was released (or never acquired) - any other state is in use. (A slot // index comparison against _leak_tag_free_count proves nothing here: the // free list is a LIFO stack of indices, not an index-bounded region.) - // Read under the pool lock (see acquireLeakTag()'s comment): the intended - // caller (ReferenceChainTracker's BFS poll thread) holds no table lock in - // its polling path, and without this lock the releaseLeakTag() zeroing on - // the GC-callback thread would race these reads. - _leak_tag_pool_lock.lock(); - bool in_use = _leak_tag_info[idx].call_trace_id != 0 || - _leak_tag_info[idx].tid != 0; + bool in_use = + !(_leak_tag_info[idx].call_trace_id == 0 && _leak_tag_info[idx].tid == 0); if (in_use) { *out_call_trace_id = _leak_tag_info[idx].call_trace_id; *out_tid = _leak_tag_info[idx].tid; @@ -515,6 +515,19 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, u32 age; int distinct_ages; // age diversity of this entry's tid (computed below) bool leak_tag_recorded; // record already holds a pool tag (reused below) + // Snapshot payload (phase 1, under the shared lock): the tagging state + // machine below runs OUTSIDE _table_lock, so everything it reads from + // the record must be copied here - table slots move under cleanup_'s + // compaction and their payloads change, but these snapshots (plus the + // local ref) are stable for the duration of the poll. + jobject ref; // NewLocalRef of the entry's object (phase 1) + u64 call_trace_id; + u32 cached_klass_id; + u64 alloc_size; + u64 entry_time; // identity for the phase-3 write-back check + jlong recorded_leak_tag; + jlong write_leak_tag; // phase-3 result + bool write_back; }; // Stack scratch: matching entries are bounded by the tracking table's // small live population (~hundreds); tag in scan order beyond capacity. @@ -556,25 +569,38 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, // Check if this entry's (class, allocating thread) matches any // candidate: class alone is not enough - only the instances a // candidate's QUALIFYING tids allocated are in tagging scope. + // + // Klass-id pre-filter first: the scan runs over the ENTIRE table (up to + // MAX_TRACKING_TABLE_SIZE = 262144 entries) on the ~1s poll cadence, + // and the overwhelming majority of entries match no candidate klass. + // Comparing the entry's klass_id against the <= 5 candidate ids inline + // (sorted small set, no allocation) short-circuits before the nested + // candidate x qualifying-tid loops - the old shape cost up to + // ~10M compares per poll on a full table. u32 kid = _table[i].cached_klass_id; if (kid == 0) { continue; } - bool match = false; + int matched_candidate = -1; for (int k = 0; k < candidate_count; k++) { - if (candidates[k].klass_id != kid || - candidates[k].qualifying_tid_count <= 0) { - continue; + if (candidates[k].klass_id == kid && + candidates[k].qualifying_tid_count > 0) { + matched_candidate = k; + break; } - for (int q = 0; q < candidates[k].qualifying_tid_count; q++) { - if (candidates[k].qualifying_tids[q] == _table[i].tid) { + } + if (matched_candidate < 0) { + continue; + } + bool match = false; + { + const KlassCandidate &kc = candidates[matched_candidate]; + for (int q = 0; q < kc.qualifying_tid_count; q++) { + if (kc.qualifying_tids[q] == _table[i].tid) { match = true; break; } } - if (match) { - break; - } } if (!match) { continue; @@ -584,14 +610,34 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, // or the search restarts, and the state machine below must re-act on // the CURRENT JVMTI tag (re-establish, correlate, or leave alone). if (n_candidates < (int)(sizeof(scratch) / sizeof(scratch[0]))) { - scratch[n_candidates].table_idx = i; - scratch[n_candidates].tid = _table[i].tid; - scratch[n_candidates].age = _table[i].age; - scratch[n_candidates].distinct_ages = 0; - scratch[n_candidates].leak_tag_recorded = _table[i].leak_tag != 0; + TagCandidate &sc = scratch[n_candidates]; + sc.table_idx = i; + sc.tid = _table[i].tid; + sc.age = _table[i].age; + sc.distinct_ages = 0; + sc.recorded_leak_tag = _table[i].leak_tag; + sc.leak_tag_recorded = sc.recorded_leak_tag != 0; + sc.call_trace_id = _table[i].call_trace_id; + sc.cached_klass_id = _table[i].cached_klass_id; + sc.alloc_size = _table[i].alloc._size; + sc.entry_time = _table[i].time; + sc.write_leak_tag = 0; + sc.write_back = false; + // Local ref taken NOW, under the shared lock: past this point the + // record's ref can be reaped by a concurrent exclusive sweep, and the + // JVMTI tagging below needs a live jobject. + sc.ref = env->NewLocalRef(_table[i].ref); n_candidates++; } } + // Phase 2 runs unlocked: everything below reads only the scratch + // snapshots and object-level JVMTI/JNI state - no _table access - so the + // shared lock (which blocks the exclusive cleanup/flush sweeps) is + // released before the GetTag/SetTag/correlate work. This bounds the + // sweep-stall the old hold-across-JVMTI shape caused: the tagging + // machine ran NewLocalRef + jvmti->GetTag + jvmti->SetTag + a + // cross-singleton correlate call per candidate, all inside lockShared. + _table_lock.unlockShared(); // Compute per-tid distinct surviving ages (matching entries only - the // same diversity signal the epoch fold uses for clustering, computed here // directly from the tracked entries so the ranking reflects exactly the @@ -671,11 +717,12 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, // - no tag: plain SetTag (first tagging, or re-establishing after a // search restart wiped all tags via releaseSearchTags()). for (int c = 0; c < n_candidates; c++) { - u32 i = scratch[c].table_idx; - jobject ref = env->NewLocalRef(_table[i].ref); + TagCandidate &sc = scratch[c]; + jobject ref = sc.ref; if (ref == nullptr) { - // Object was collected between the null check and now - its record's - // tag (if any) is released by the GC cleanup path, nothing to do. + // Object was collected between the phase-1 NewLocalRef and now - its + // record's tag (if any) is released by the GC cleanup path, nothing + // to do. continue; } jlong existing = 0; @@ -685,61 +732,36 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, if (tag_err == JVMTI_ERROR_NONE && existing >= LEAK_TAG_BASE) { // Already carries a leak tag (ours, or one adopted below) - waiting // for the BFS interception. Make sure the record remembers it. - leak_tag = _table[i].leak_tag != 0 ? _table[i].leak_tag : existing; + leak_tag = sc.recorded_leak_tag != 0 ? sc.recorded_leak_tag : existing; } else if (tag_err == JVMTI_ERROR_NONE && existing > 0) { // Frontier tag: the BFS already admitted this object. Correlate the // existing entry rather than retagging - see the block comment above. - // - // Lock order (enforced, asymmetric): this call runs under THIS tracker's - // SHARED _table_lock and takes ReferenceChainTracker-internal locks - // (FrontierTable's SpinLock - held only inside individual - // insert/lookup/setLeakTag method bodies - and _resolved_chains_lock - // inside invalidateResolvedChain()). Every direction that could invert - // this is excluded today: RCT entry points into LT - // (resolveCandidateRepresentative, selectLeakCandidates, - // tagLeakInstances from pollWatchedTargets) take the table lock before - // touching any shared tracker state and are called with no RCT-internal - // lock held, so no path acquires _table_lock while already holding an - // RCT-internal lock. Any new RCT -> LT call made while holding an - // RCT-internal lock would invert the order and deadlock against this - // site. (Moving the correlate call outside the shared-lock section was - // rejected: it needs _table[i].ref/cached_klass_id, and the table - // index/weak ref can be compacted or reaped by a concurrent - // cleanup_table() once the shared lock is released - re-touching them - // unlocked would read freed/moved slots.) - leak_tag = _table[i].leak_tag; + leak_tag = sc.recorded_leak_tag; if (leak_tag == 0) { - leak_tag = acquireLeakTag(_table[i].call_trace_id, _table[i].tid); + leak_tag = acquireLeakTag(sc.call_trace_id, sc.tid); if (leak_tag == 0) { env->DeleteLocalRef(ref); + sc.ref = nullptr; continue; // pool exhausted - other candidates may still correlate } } if (!ReferenceChainTracker::instance()->correlateAdmittedLeakTag( - existing, leak_tag, _table[i].cached_klass_id)) { + existing, leak_tag, sc.cached_klass_id)) { // Not a live frontier tag after all (search just restarted) - // fall back to plain tagging. need_set = true; } - } else if (tag_err == JVMTI_ERROR_NONE && existing < 0) { - // Negative tag: a stable class tag from the shared class-tag allocator - // (classTagAllocator.h) - the candidate is a java.lang.Class mirror. - // Replacing it with a positive leak tag would break class resolution - // and frontier matching for the represented class. Preserve the - // installed class tag untouched; the entry's pool-tag bookkeeping - // stays as it is. - leak_tag = _table[i].leak_tag; - need_set = false; } else { // No tag: first tagging, or re-establishment after a restart wiped // all tags (releaseSearchTags() clears every JVMTI tag while the // record keeps its pool tag - reusing it keeps pool accounting // stable across restarts). - leak_tag = _table[i].leak_tag; + leak_tag = sc.recorded_leak_tag; if (leak_tag == 0) { - leak_tag = acquireLeakTag(_table[i].call_trace_id, _table[i].tid); + leak_tag = acquireLeakTag(sc.call_trace_id, sc.tid); if (leak_tag == 0) { env->DeleteLocalRef(ref); + sc.ref = nullptr; break; // pool exhausted } } @@ -748,26 +770,18 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, if (need_set) { jvmti->SetTag(ref, leak_tag); } - // Store under the SHARED table lock: the only other writers are exclusive- - // lock holders (cleanup_table() zeroing a dead entry, start() reclaiming - // owned tags), which a shared holder excludes - and tagLeakInstances() - // itself runs on the single BFS poll thread, so no second shared-lock - // writer exists. Readers under either lock mode therefore always see a - // value from one of those serialized writers (aligned jlong stores are - // atomic on all supported architectures); a shared-mode reader may observe - // a stale 0 and re-tag - the correlate/re-establish state machine above is - // idempotent for that case. If a second shared-lock scanner ever starts - // writing leak_tag, this store (and those reads) must move under the - // exclusive lock or become atomic. - _table[i].leak_tag = leak_tag; + // Record write-back is deferred to phase 3 (the table lock is not held + // here); the summary below consumes the snapshot fields. + sc.write_leak_tag = leak_tag; + sc.write_back = true; tagged++; // Accumulate into the per-poll summary above instead of logging per // instance - one stable-pool poll re-logged all 256 tags' identical // lines every 1.4s before this. - u32 tagged_kid = _table[i].cached_klass_id; - jint tagged_tid = _table[i].tid; - u64 tagged_age = _table[i].age; - u64 tagged_size = _table[i].alloc._size; + u32 tagged_kid = sc.cached_klass_id; + jint tagged_tid = sc.tid; + u64 tagged_age = sc.age; + u64 tagged_size = sc.alloc_size; int g = 0; while (g < summary_count && (summary[g].klass_id != tagged_kid || summary[g].tid != tagged_tid)) { @@ -800,9 +814,36 @@ int LivenessTracker::tagLeakInstances(jvmtiEnv *jvmti, summary[g].max_size = tagged_size; } } - env->DeleteLocalRef(ref); + } + // Phase 3: write the resulting tags back to the records under a short + // shared-lock pass. A slot is only written when it still holds the SAME + // entry (identity = publish flag + allocation time + call_trace_id): a + // concurrent exclusive sweep may have reaped it (compaction moves slots), + // and writing a moved entry's tag would corrupt whoever reuses the slot. + // A skipped write-back is self-healing: the object carries the tag, the + // next poll's adoption branch re-records it, and start()'s pool rebuild + // reclaims any tag whose record is gone for good. + _table_lock.lockShared(); + for (int c = 0; c < n_candidates; c++) { + TagCandidate &sc = scratch[c]; + if (!sc.write_back) { + continue; + } + if (__atomic_load_n(&_table[sc.table_idx].ready, __ATOMIC_ACQUIRE) == 1 && + _table[sc.table_idx].time == sc.entry_time && + _table[sc.table_idx].call_trace_id == sc.call_trace_id) { + _table[sc.table_idx].leak_tag = sc.write_leak_tag; + } } _table_lock.unlockShared(); + // Release the phase-1 local refs (the state machine nulled ref on its own + // early-exit paths; those are already deleted). + for (int c = 0; c < n_candidates; c++) { + if (scratch[c].ref != nullptr) { + env->DeleteLocalRef(scratch[c].ref); + scratch[c].ref = nullptr; + } + } for (int g = 0; g < summary_count; g++) { TEST_LOG("LivenessTracker::tagLeakInstances summary klass_id=%u " "tid=%d tagged=%d need_set=%d min_age=%llu max_age=%llu " @@ -865,6 +906,45 @@ void LivenessTracker::insertThreadGen(KlassCountScratch &scratch, // the oldest[] array still captures instances from all threads. } +void LivenessTracker::klassCountScratchReset() { + _klass_count_scratch_size = 0; + memset(_klass_count_index, 0, sizeof(_klass_count_index)); +} + +LivenessTracker::KlassCountScratch *LivenessTracker::klassCountScratchSlot( + u32 klass_id, bool allocate) { + // Same mix() as the frontier's slot hashing (referenceChains.h); the + // klass_id need not be a tag, only uniformly distributed. + u64 i = (u64)(klass_id * 0x9E3779B97F4A7C15ULL) >> (64 - 9); + i &= KLASS_COUNT_INDEX_SLOTS - 1; + while (true) { + u16 slot = _klass_count_index[i]; + if (slot == 0) { + if (!allocate || + _klass_count_scratch_size >= MAX_KLASS_POPULATION_ENTRIES) { + return nullptr; + } + int idx = _klass_count_scratch_size++; + _klass_count_index[i] = (u16)(idx + 1); + KlassCountScratch &fresh = _klass_count_scratch[idx]; + // Full re-initialization: slots are reused across epochs (only the + // size + index are reset), so every field the fold reads must be + // cleared here, not just at construction. + fresh.klass_id = klass_id; + fresh.ages_count = 0; + fresh.ages_saturated = false; + fresh.oldest_count = 0; + fresh.thread_count = 0; + return &fresh; + } + KlassCountScratch &entry = _klass_count_scratch[slot - 1]; + if (entry.klass_id == klass_id) { + return &entry; + } + i = (i + 1) & (KLASS_COUNT_INDEX_SLOTS - 1); + } +} + void LivenessTracker::accumulateKlassCount(u32 klass_id, jlong age, jweak sample_source, jint tid) { @@ -873,45 +953,44 @@ void LivenessTracker::accumulateKlassCount(u32 klass_id, jlong age, // number of unique age values. This is the "generation // count" — if new instances keep arriving while old ones // survive, the number of distinct ages grows. - for (int i = 0; i < _klass_count_scratch_size; i++) { - if (_klass_count_scratch[i].klass_id == klass_id) { - auto &entry = _klass_count_scratch[i]; - // Per-class age dedup: only count each age once for the klass' - // generation count. But per-site tracking and oldest[] must see - // EVERY surviving object, not just the first per age — so those - // run unconditionally below, outside this dedup check. - bool age_seen = false; - for (u32 a : entry.ages) { - if (a == (u32)age) { - age_seen = true; - break; - } + // Direct-indexed (klassCountScratchSlot): the caller holds the exclusive + // table lock and this runs once per surviving table entry, so the old + // linear scan over _klass_count_scratch scaled O(entries x 256). + KlassCountScratch *entry = klassCountScratchSlot(klass_id, true); + if (entry != nullptr) { + // Per-class age dedup: only count each age once for the klass' + // generation count. But per-site tracking and oldest[] must see + // EVERY surviving object, not just the first per age — so those + // run unconditionally below, outside this dedup check. + bool age_seen = false; + for (int a = 0; a < entry->ages_count; a++) { + if (entry->ages[a] == (u32)age) { + age_seen = true; + break; } - if (!age_seen) { - entry.ages.push_back((u32)age); + } + if (!age_seen) { + if (entry->ages_count < KlassCountScratch::MAX_DISTINCT_AGES) { + entry->ages[entry->ages_count++] = (u32)age; + } else { + // Saturated: a klass this rich in distinct surviving ages in one + // epoch is already the strongest possible generation-count signal; + // stop growing (and stop paying the dedup scan) rather than + // allocating. See MAX_DISTINCT_AGES's comment. + entry->ages_saturated = true; } - // Track top-N oldest instances (Lindy bias): insert this sample - // into the oldest[] array, sorted by age descending, capped at - // MAX_OLDEST_SAMPLES. Runs for every object, not just new ages. - insertOldestSample(entry, sample_source, (u32)age, tid); - // Track per-thread distinct surviving generations (Cork/Swat - // heuristic): add this object's age to its thread's age set. - // Runs for every object — the thread's generation cardinality is - // the leak signal, and it must see all surviving objects to be - // accurate. - insertThreadGen(entry, tid, (u32)age); - return; } - } - if (_klass_count_scratch_size < MAX_KLASS_POPULATION_ENTRIES) { - KlassCountScratch &slot = _klass_count_scratch[_klass_count_scratch_size++]; - slot.klass_id = klass_id; - slot.ages.clear(); - slot.ages.push_back((u32)age); - slot.oldest_count = 0; - slot.thread_count = 0; - insertOldestSample(slot, sample_source, (u32)age, tid); - insertThreadGen(slot, tid, (u32)age); + // Track top-N oldest instances (Lindy bias): insert this sample + // into the oldest[] array, sorted by age descending, capped at + // MAX_OLDEST_SAMPLES. Runs for every object, not just new ages. + insertOldestSample(*entry, sample_source, (u32)age, tid); + // Track per-thread distinct surviving generations (Cork/Swat + // heuristic): add this object's age to its thread's age set. + // Runs for every object — the thread's generation cardinality is + // the leak signal, and it must see all surviving objects to be + // accurate. + insertThreadGen(*entry, tid, (u32)age); + return; } // else: this epoch's scratch snapshot already holds // MAX_KLASS_POPULATION_ENTRIES distinct surviving klasses - klass_id's @@ -972,6 +1051,9 @@ jweak LivenessTracker::recordKlassPopulationSampleLocked( _klass_population[slot].ring_fill = 0; _klass_population[slot].consecutive_positive = 0; _klass_population[slot].cached_slope = 0.0; + // A fresh/evicted slot has never been probed - force the staleness + // check on this fold. + _klass_population[slot].last_rep_probe_epoch = 0; // A reused (evicted) slot's previous class's per-tid trends must not // leak onto the new one, same as the fields above. _klass_population[slot].tid_trend_count = 0; @@ -1021,30 +1103,8 @@ void LivenessTracker::mintStableClassTagIfNeeded(JNIEnv *env, int slot, jlong tag = 0; if (jvmti->GetTag(klass, &tag) == JVMTI_ERROR_NONE) { if (tag == 0) { - jlong new_tag = ClassTagAllocator::next(); - if (jvmti->SetTag(klass, new_tag) == JVMTI_ERROR_NONE) { - // Adopt the tag actually installed on the class object: the - // reference-chain tracker's resolveLoadedClasses() may have - // installed its own tag between our GetTag and SetTag (two SetTag - // calls on the same untagged class - the last writer wins on the - // class object). Re-read so both trackers keep the ONE tag the - // class carries; publishing our own minted tag otherwise would - // permanently disconnect this entry's leak correlation from the - // class tag the reference-chain side keys off of. - jlong installed = 0; - if (jvmti->GetTag(klass, &installed) == JVMTI_ERROR_NONE && - installed != 0) { - tag = installed; - } else { - tag = new_tag; - } - } else { - // SetTag failed: publish nothing. stable_class_tag stays 0 and - // the next fold's minting retry (the need_mint stale-representative - // probe) calls this method again - a tag that is not on the class - // object must never be cached as the stable tag. - tag = 0; - } + tag = ClassTagAllocator::next(); + jvmti->SetTag(klass, tag); } _klass_population[slot].stable_class_tag = tag; } @@ -1062,7 +1122,7 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, KlassCountScratch &s = _klass_count_scratch[i]; TEST_LOG("LivenessTracker::foldKlassCountsLocked scratch[%d] klass_id=%u gen_count=%zu " "thread_count=%d oldest_count=%d", - i, s.klass_id, s.ages.size(), s.thread_count, s.oldest_count); + i, s.klass_id, s.ages_count, s.thread_count, s.oldest_count); for (int ti = 0; ti < s.thread_count; ti++) { TEST_LOG(" thread[%d] tid=%d age_count=%u", ti, (int)s.threads[ti].tid, s.threads[ti].age_count); @@ -1071,7 +1131,7 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, bool created; jweak evicted[KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS]; int evicted_count = 0; - recordKlassPopulationSampleLocked(s.klass_id, (u32)s.ages.size(), + recordKlassPopulationSampleLocked(s.klass_id, (u32)s.ages_count, epoch, &slot, &created, evicted, &evicted_count, KlassPopulationEntry::MAX_REPRESENTATIVES_PER_KLASS); @@ -1110,21 +1170,35 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, bool need_mint = created || _klass_population[slot].representative_count == 0; if (!need_mint) { - // Check if all representatives are stale - bool any_live = false; - for (int r = 0; r < _klass_population[slot].representative_count; r++) { - jweak rep = _klass_population[slot].representatives[r]; - if (rep != nullptr) { - jobject probe = env->NewLocalRef(rep); - if (probe != nullptr) { - any_live = true; + // Staleness probe amortization: a jweak's pointer never nulls on + // collection, so the only way to detect a dead representative is a + // NewLocalRef probe - JNI churn under the exclusive table lock when + // run for every klass every epoch. A live representative stays live + // until collected, and a missed collection costs one epoch with a + // stale rep (the minting retry picks it up on the next probe), so + // probing every REP_PROBE_EPOCH_INTERVAL epochs bounds the gap + // without paying the JNI round-trips per sweep. + constexpr u64 REP_PROBE_EPOCH_INTERVAL = 4; + bool probe_due = + epoch - _klass_population[slot].last_rep_probe_epoch >= + REP_PROBE_EPOCH_INTERVAL; + if (probe_due) { + _klass_population[slot].last_rep_probe_epoch = epoch; + bool any_live = false; + for (int r = 0; r < _klass_population[slot].representative_count; r++) { + jweak rep = _klass_population[slot].representatives[r]; + if (rep != nullptr) { + jobject probe = env->NewLocalRef(rep); + if (probe != nullptr) { + any_live = true; + env->DeleteLocalRef(probe); + break; + } env->DeleteLocalRef(probe); - break; } - env->DeleteLocalRef(probe); } + need_mint = !any_live; } - need_mint = !any_live; } // Compute the dominant allocating thread (highest generation // cardinality — most distinct surviving GC ages). This reuses @@ -1193,19 +1267,6 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, jobject strong = env->NewLocalRef(s.oldest[r].ref); if (strong != nullptr) { jweak rep = env->NewWeakGlobalRef(strong); - if (rep == nullptr) { - // NewWeakGlobalRef failed under memory pressure (an - // OutOfMemoryError may be pending). Store no representative - // and stop minting: continuing JNI calls with a pending - // exception is undefined behavior, and a null rep would - // corrupt representative_count. The next epoch's need_mint - // retry re-attempts minting. - if (env->ExceptionCheck()) { - env->ExceptionClear(); - } - env->DeleteLocalRef(strong); - break; - } int idx = _klass_population[slot].representative_count++; _klass_population[slot].representatives[idx] = rep; _klass_population[slot].rep_tids[idx] = dominant_tid; @@ -1225,15 +1286,6 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, jobject strong = env->NewLocalRef(s.oldest[r].ref); if (strong != nullptr) { jweak rep = env->NewWeakGlobalRef(strong); - if (rep == nullptr) { - // Same OOM handling as the dominant-thread loop above: no null - // representative, pending exception cleared, minting stops. - if (env->ExceptionCheck()) { - env->ExceptionClear(); - } - env->DeleteLocalRef(strong); - break; - } int idx = _klass_population[slot].representative_count++; _klass_population[slot].representatives[idx] = rep; _klass_population[slot].rep_tids[idx] = s.oldest[r].tid; @@ -1253,7 +1305,7 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, minted, s.klass_id, (int)dominant_tid, dominant_gens); } } - _klass_count_scratch_size = 0; + klassCountScratchReset(); // Zero-sample pass: a klass whose every tracked instance died this epoch // never appears in _klass_count_scratch, so the loop above never refreshes @@ -1286,8 +1338,8 @@ void LivenessTracker::foldKlassCountsLocked(JNIEnv *env, u64 epoch, } bool LivenessTracker::hasQualifyingGrowth(const KlassPopulationEntry &entry) const { - RingThirdsStats stats; - if (!ringThirdsStats( + RingWindowStats stats; + if (!ringWindowStats( entry.ring_head, entry.ring_fill, KLASS_POPULATION_RING_SIZE, KLASS_POPULATION_MIN_FILL_FOR_TREND, [&entry](int i) { return (double)entry.count_ring[i]; }, &stats)) { @@ -1322,8 +1374,8 @@ bool LivenessTracker::hasQualifyingGrowth(const KlassPopulationEntry &entry) con bool LivenessTracker::hasQualifyingTidGrowth( const KlassPopulationEntry::TidTrend &trend) const { - RingThirdsStats stats; - if (!ringThirdsStats( + RingWindowStats stats; + if (!ringWindowStats( trend.ring_head, trend.ring_fill, KlassPopulationEntry::TID_TREND_RING_SIZE, TID_TREND_MIN_FILL_FOR_TREND, @@ -1514,8 +1566,8 @@ bool LivenessTracker::heapFloorRising() const { // loadAcquire() here is what makes the payload writes below visible. u8 fill = loadAcquire(_heap_floor_ring_fill); u8 head = loadAcquire(_heap_floor_ring_head); - RingThirdsStats stats; - if (!ringThirdsStats( + RingWindowStats stats; + if (!ringWindowStats( head, fill, KLASS_POPULATION_RING_SIZE, KLASS_POPULATION_MIN_FILL_FOR_TREND, [this](int i) { return (double)load(_heap_floor_ring[i]); }, @@ -1587,8 +1639,8 @@ double LivenessTracker::secondsToOOM() const { (int)fill, KLASS_POPULATION_MIN_FILL_FOR_TREND); return -1; } - RingThirdsStats time_stats; - if (!ringThirdsStats( + RingWindowStats time_stats; + if (!ringWindowStats( head, fill, KLASS_POPULATION_RING_SIZE, KLASS_POPULATION_MIN_FILL_FOR_TREND, [this](int i) { return (double)load(_heap_floor_time_ring[i]); }, @@ -1606,45 +1658,58 @@ double LivenessTracker::secondsToOOM() const { const char *best_source = "none"; double best_recent_mean = 0; - RingThirdsStats heap_bytes; + RingWindowStats heap_bytes; if (max_heap > 0 && - ringThirdsStats(head, fill, KLASS_POPULATION_RING_SIZE, + ringWindowStats(head, fill, KLASS_POPULATION_RING_SIZE, KLASS_POPULATION_MIN_FILL_FOR_TREND, [this](int i) { return (double)load(_heap_floor_ring[i]); }, &heap_bytes) && corroborateRecentHalf(head, fill, KLASS_POPULATION_RING_SIZE, HEAP_FLOOR_RECENT_HALF_MIN_FILL, [this](int i) { return (double)load(_heap_floor_ring[i]); })) { - double remaining = (double)max_heap - heap_bytes.recent_mean; - double secs = remaining <= 0 - ? 0 - : (remaining * time_delta_ns) / - (heap_bytes.recent_mean - heap_bytes.earliest_mean) / 1e9; - if (best_seconds < 0 || secs < best_seconds) { - best_seconds = secs; - best_source = "heap"; - best_recent_mean = heap_bytes.recent_mean; + // Denominator is the full-window byte delta (endpoints of the fitted + // line). A dip-then-recover window can pass corroborateRecentHalf() + // (its recent half rises) while the full-window endpoints agree, making + // this delta ~0 - an unguarded division projects +inf seconds, which + // silently disables the urgency ramp from this boundary forever (best + // -seconds would be set to +inf and never beaten). Skip the boundary + // instead: no usable rate, no projection. + double heap_delta = heap_bytes.recent_mean - heap_bytes.earliest_mean; + if (heap_delta > 0) { + double remaining = (double)max_heap - heap_bytes.recent_mean; + double secs = remaining <= 0 + ? 0 + : (remaining * time_delta_ns) / heap_delta / 1e9; + if (best_seconds < 0 || secs < best_seconds) { + best_seconds = secs; + best_source = "heap"; + best_recent_mean = heap_bytes.recent_mean; + } } } - RingThirdsStats container_bytes; + RingWindowStats container_bytes; if (container_limit > 0 && - ringThirdsStats(head, fill, KLASS_POPULATION_RING_SIZE, + ringWindowStats(head, fill, KLASS_POPULATION_RING_SIZE, KLASS_POPULATION_MIN_FILL_FOR_TREND, [this](int i) { return (double)load(_container_mem_ring[i]); }, &container_bytes) && corroborateRecentHalf(head, fill, KLASS_POPULATION_RING_SIZE, HEAP_FLOOR_RECENT_HALF_MIN_FILL, [this](int i) { return (double)load(_container_mem_ring[i]); })) { - double remaining = (double)container_limit - container_bytes.recent_mean; - double secs = remaining <= 0 - ? 0 - : (remaining * time_delta_ns) / - (container_bytes.recent_mean - container_bytes.earliest_mean) / 1e9; - if (best_seconds < 0 || secs < best_seconds) { - best_seconds = secs; - best_source = "container"; - best_recent_mean = container_bytes.recent_mean; + // Same zero-denominator guard as the heap boundary above. + double container_delta = + container_bytes.recent_mean - container_bytes.earliest_mean; + if (container_delta > 0) { + double remaining = (double)container_limit - container_bytes.recent_mean; + double secs = remaining <= 0 + ? 0 + : (remaining * time_delta_ns) / container_delta / 1e9; + if (best_seconds < 0 || secs < best_seconds) { + best_seconds = secs; + best_source = "container"; + best_recent_mean = container_bytes.recent_mean; + } } } @@ -1677,11 +1742,7 @@ int LivenessTracker::selectLeakCandidates(KlassCandidate *out, int max) { // for why an aggregate, non-attributed signal can only raise or lower the // bar uniformly, never reorder candidates against each other. Lock-free // (heapFloorRising()'s own comment), so no relation to _table_lock below. - // Computed once: the trailing diagnostic TEST_LOG at the end of this - // function reuses this cached result instead of re-evaluating the full - // O(ring_fill) ring scan a second time per BFS-thread wake. - const bool heap_floor_rising = heapFloorRising(); - const int required_hysteresis = heap_floor_rising + const int required_hysteresis = heapFloorRising() ? LEAK_TREND_HYSTERESIS_CORROBORATED : LEAK_TREND_HYSTERESIS_BASE; @@ -1770,7 +1831,7 @@ int LivenessTracker::selectLeakCandidates(KlassCandidate *out, int max) { _table_lock.unlockShared(); TEST_LOG("LivenessTracker::selectLeakCandidates returning %d candidates (required_hysteresis=%d, heapFloorRising=%d)", count, required_hysteresis, - (int)heap_floor_rising); + (int)heapFloorRising()); return count; } @@ -1922,25 +1983,22 @@ void LivenessTracker::flush_table(std::set *tracked_thread_ids) { if (_table[i].cached_klass_id != 0) { // Already resolved by cleanup_table()'s survivor loop this epoch // (resolveKlassId(), only when _gc_generations is enabled) - reuse - // it instead of repeating the GetObjectClass+Class.getName()+ - // lookupClass() JNI round-trip for the same object. + // it instead of repeating the JNI round-trip for the same object. class_id = _table[i].cached_klass_id; } else { - jclass clz = env->GetObjectClass(ref); - jstring name_str = (jstring)env->CallObjectMethod(clz, _Class_getName); - env->DeleteLocalRef(clz); - jniExceptionCheck(env); - // name_str can be null if the call above threw and - // jniExceptionCheck() cleared the pending exception rather than - // propagating it - GetStringUTFChars()/ReleaseStringUTFChars() - // require a non-null jstring (mirrors resolveKlassId()'s own guard). - if (name_str != nullptr) { - const char *name = env->GetStringUTFChars(name_str, nullptr); - if (name != nullptr) { - class_id = Profiler::instance()->lookupClass(name, strlen(name)); - env->ReleaseStringUTFChars(name_str, name); - } - env->DeleteLocalRef(name_str); + // Same sequence as resolveKlassId(): GetClassSignature + + // normalizeClassSignature + lookupClass (slash-notation key), NOT + // the old Class.getName() path. The two produce DIFFERENT + // StringDictionary keys for the same class ("com/foo/Bar" vs + // "com.foo.Bar"), so a cache miss resolved through getName() could + // emit a different class id for the same class than the cache hit + // path right above - two id spaces in one event stream. resolve + // KlassId() also caches back into the entry, so a later sweep pays + // one u32 read instead of this round-trip. + int resolved = (int)resolveKlassId(env, ref); + class_id = resolved; + if (resolved > 0) { + _table[i].cached_klass_id = (u32)resolved; } } @@ -2033,6 +2091,9 @@ Error LivenessTracker::start(Arguments &args) { } } _table_lock.unlock(); + // Rebuild under the pool lock: tagLeakInstances()' unlocked tagging phase + // and getLeakTagInfo() readers may run concurrently with start(). + _leak_tag_pool_lock.lock(); int free_w = 0; for (int i = 0; i < LEAK_TAG_POOL_SIZE; i++) { if (tag_owned[i]) { @@ -2044,6 +2105,7 @@ Error LivenessTracker::start(Arguments &args) { free_w++; } _leak_tag_free_count = free_w; + _leak_tag_pool_lock.unlock(); if (!_enabled) { // disabled return Error::OK; @@ -2094,15 +2156,6 @@ Error LivenessTracker::initialize(Arguments &args) { // start gets the correct setting even when the table persists across recordings. _record_heap_usage = args._record_heap_usage; - // Fresh recording: no chase is open, so no watched tids and no urgency - // boost may leak in from a previous recording's lifecycle. This MUST run on - // EVERY start - not only the first initialization: the `_initialized` - // early return below would otherwise leave a previous recording's watched - // thread set and urgent-tracking state active, admitting the new - // recording's unrelated allocations at 100 percent. - __atomic_store_n(&_watched_tid_count, 0, __ATOMIC_RELEASE); - __atomic_store_n(&_urgent_tracking, false, __ATOMIC_RELEASE); - if (_initialized) { // if the tracker was previously initialized return the stored result for // consistency this hack also means that if the profiler is started with @@ -2168,6 +2221,11 @@ Error LivenessTracker::initialize(Arguments &args) { // decision in track() is an integer compare rather than a double multiply. _subsample = SubsampleRate(args._live_samples_ratio); + // Fresh recording: no chase is open, so no watched tids and no urgency + // boost may leak in from a previous recording's lifecycle. + __atomic_store_n(&_watched_tid_count, 0, __ATOMIC_RELEASE); + __atomic_store_n(&_urgent_tracking, false, __ATOMIC_RELEASE); + _table_size = 0; _table_cap = std::min(2048, _table_max_cap); // with default 512k sampling interval, it's @@ -2212,7 +2270,28 @@ static ThreadLocal skipped; // per thread and probabilistic by design). bool LivenessTracker::admitForTracking(jint tid) { if (__atomic_load_n(&_urgent_tracking, __ATOMIC_ACQUIRE)) { - return true; + // Volume backstop for the urgency boost (100% admission, bypassing even + // the subsample draw): once the table is at its high-water mark the + // boost's own cost - track()'s overflow branch firing a forced sweep on + // the SampledObjectAlloc callback stack, per admission - outweighs the + // extra diagnostic coverage. Fall back to the watched-tid-only boost, + // which keeps 100% admission exactly where the leak data value is (the + // candidate sites) while every other thread goes back to the configured + // subsample ratio. Admissions at the high-water mark are still counted + // so the degradation is observable in production. + if (_table_max_cap > 0 && _table_size >= _table_max_cap) { + Counters::increment(LIVENESS_URGENT_BOOST_BACKED_OFF); + int n = __atomic_load_n(&_watched_tid_count, __ATOMIC_ACQUIRE); + for (int i = 0; i < n; i++) { + if (_watched_tids[i] == tid) { + return true; + } + } + // Fall through to the configured-ratio draw below. + } else { + Counters::increment(LIVENESS_URGENT_BOOST_ADMITS); + return true; + } } // Count+array two-phase publish (noteSelectedCandidates() writes the // slots before release-storing the count): the acquire load pairs with @@ -2221,7 +2300,7 @@ bool LivenessTracker::admitForTracking(jint tid) { // for why RELAXED is not an option on arm64. int n = __atomic_load_n(&_watched_tid_count, __ATOMIC_ACQUIRE); for (int i = 0; i < n; i++) { - if (__atomic_load_n(&_watched_tids[i], __ATOMIC_RELAXED) == tid) { + if (_watched_tids[i] == tid) { return true; } } @@ -2275,15 +2354,12 @@ void LivenessTracker::noteSelectedCandidates(const KlassCandidate *candidates, } full: // Copy into the live array before publishing the count (two-phase - // publish mirrored by admitForTracking()'s acquire load). Slot accesses - // are ATOMIC (relaxed): the poll thread rewrites slots while allocation - // threads can still read them under an acquire count load that observed - // the OLD count - plain loads/stores there are a C++ data race (UB). - // A reader mid-scan may transiently mix old and new slot values below the - // OLD count - harmless: admission is advisory, and the worst case is one + // publish mirrored by admitForTracking()'s acquire load). A reader + // mid-scan may transiently mix old and new slot values below the OLD + // count - harmless: admission is advisory, and the worst case is one // allocation admitted per the previous poll's set. for (int i = 0; i < n; i++) { - __atomic_store_n(&_watched_tids[i], tids[i], __ATOMIC_RELAXED); + _watched_tids[i] = tids[i]; } __atomic_store_n(&_watched_tid_count, n, __ATOMIC_RELEASE); if (n > 0) { @@ -2293,8 +2369,7 @@ void LivenessTracker::noteSelectedCandidates(const KlassCandidate *candidates, TEST_LOG("LivenessTracker::noteSelectedCandidates watched tids[%d]:", n); for (int i = 0; i < n; i++) { - TEST_LOG(" watched tid=%d", - __atomic_load_n(&_watched_tids[i], __ATOMIC_RELAXED)); + TEST_LOG(" watched tid=%d", _watched_tids[i]); } } } @@ -2349,7 +2424,18 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, } bool retried = false; retry: - if (!_table_lock.tryLockShared()) { + // EXCLUSIVE, not shared: the fill below RE-USES slots (idx < _table_cap + // after _table_size wraps past reaped entries), and a shared lock does + // not exclude other shared holders - a scanner (tagLeakInstances(), + // getLiveTraceIds()) that had already passed its ready==1 check for the + // slot's OLD entry could read the payload mid-fill and get a torn mix of + // old and new values. The unpublish/publish dance cannot close that + // (the scanner's check happened before the unpublish). Exclusive here + // serializes fills against scanners; contention is bounded because this + // lock is only taken once per SAMPLED allocation (admitForTracking()'s + // subsample draw above rejects the unsampled majority before any lock + // is taken). + if (!_table_lock.tryLock()) { // we failed to add the weak reference to the table so it won't get cleaned // up otherwise env->DeleteWeakGlobalRef(ref); @@ -2385,7 +2471,7 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, __atomic_store_n(&_table[idx].ready, 1, __ATOMIC_RELEASE); } - _table_lock.unlockShared(); + _table_lock.unlock(); if (idx == _table_cap) { if (!retried) { @@ -2396,7 +2482,13 @@ void LivenessTracker::track(JNIEnv *env, AllocEvent &event, jint tid, // space. allow_resolve=false: this runs synchronously on the // allocation-sampling callback stack (see cleanup_table()'s own header // comment for why resolveKlassId() is unsafe here). - cleanup_table(true, false); + // account_epoch=false: pure reaper - no epoch claim, no survivor + // aging, no population fold. The fold's nested per-(klass,tid) loops + // ran on this hot callback stack under the exclusive table lock; + // the background/GC sweeps (which claim the epoch) do the full + // accounting, so the overflow path only pays for the reaping it + // actually needs. + cleanup_table(true, false, false); if (_table_cap < _table_max_cap) { diff --git a/ddprof-lib/src/main/cpp/livenessTracker.h b/ddprof-lib/src/main/cpp/livenessTracker.h index a942d66479..136e599609 100644 --- a/ddprof-lib/src/main/cpp/livenessTracker.h +++ b/ddprof-lib/src/main/cpp/livenessTracker.h @@ -101,9 +101,8 @@ typedef struct KlassPopulationEntry { // without this counter it gets reported as a leak candidate almost as // often as a real leak does. u8 consecutive_positive; - // Slope (regression value at the ring's newest sample minus the value at - // its oldest sample - see ringThirdsStats, livenessTracker.cpp) as of the - // last push, computed and cached by hasQualifyingGrowth() alongside + // Slope (recent third's mean minus earliest third's mean) as of the last + // push, computed and cached by hasQualifyingGrowth() alongside // consecutive_positive above - selectLeakCandidates() reads this directly // for ranking instead of re-scanning the ring: the ring only changes on // push, so a second scan at scan time would just recompute the same @@ -113,6 +112,13 @@ typedef struct KlassPopulationEntry { // existed. mutable double cached_slope; u64 last_updated_epoch; // _gc_epoch value as of the last write, for LRU + // Epoch of the last representative-staleness JNI probe (foldKlassCounts + // Locked()'s NewLocalRef check of each stored jweak). The probe is + // amortized: without this field it ran for every klass with survivors on + // EVERY epoch (up to 256 klasses x 3 jweaks of JNI churn per sweep) even + // when a representative is long-lived and nothing changed. See + // REP_PROBE_EPOCH_INTERVAL. + u64 last_rep_probe_epoch; // eviction when the table is full // Stable per-class identifier, from the process-wide, negative-tag // allocator shared with ReferenceChainTracker (classTagAllocator.h) - NOT @@ -220,8 +226,8 @@ class alignas(alignof(SpinLock)) LivenessTracker { // a klass starts being tracked." constexpr static int KLASS_POPULATION_MIN_FILL_FOR_TREND = 10; // Per-tid trend gate's own minimum fill (see KlassPopulationEntry:: - // TidTrend): 6 samples of the 16-slot ring leaves a 2-3 sample - // regression, small enough that a per-tid qualification (6 pushes + + // TidTrend): 6 samples of the 16-slot ring leaves a 2-3 sample thirds + // comparison, small enough that a per-tid qualification (6 pushes + // the 3-5 hysteresis epochs, ~9-11) completes no later than the klass // gate's own ~13-15-epoch latency, keeping the klass ring the sole // latency driver for candidate emergence. @@ -268,11 +274,11 @@ class alignas(alignof(SpinLock)) LivenessTracker { // surviving instance count: a klass whose survivors keep spanning more // distinct allocation cohorts over time is one where old instances are // not dying as new ones arrive, which is the leak shape this gate looks - // for. The gate itself is a single condition - the ring's regression end - // value must exceed its start value by a meaningful + // for. The gate itself is a single condition - the recent third's mean + // generation count must exceed the earliest third's by a meaningful // margin (LEAK_GROWTH_REL_MIN/LEAK_GROWTH_ABS_MIN, whichever is larger). - // An earlier revision of this gate also required the recent half's - // *minimum* to exceed the full window's minimum (a floor-rise check, + // An earlier revision of this gate also required the recent third's + // *minimum* to exceed the earliest third's minimum (a floor-rise check, // to reject oscillations whose peak alone passes the growth test) - that // check was tuned for raw population counts (which can run into the // thousands) and does not transfer to generation counts, which are small @@ -300,7 +306,7 @@ class alignas(alignof(SpinLock)) LivenessTracker { constexpr static int LEAK_TREND_HYSTERESIS_CORROBORATED = 3; // --- Aggregate post-GC heap floor (heapFloorRising() below) --- - // Same "regression growth + floor rise" shape as the per-klass test + // Same "mean-of-thirds growth + floor rise" shape as the per-klass test // above, applied to a single global ring of post-GC live heap size // instead of one klass's sampled population - see this class's own ring // (_heap_floor_ring below). Its own thresholds are deliberately looser @@ -330,10 +336,6 @@ class alignas(alignof(SpinLock)) LivenessTracker { // comment below). _watched_tids is two-phase-published: slots first, then // _watched_tid_count with RELEASE (admitForTracking()'s ACQUIRE load pairs // with it) - a reader never trusts a slot beyond the count it observed. - // Every slot access is ATOMIC (__atomic_* builtins, relaxed): the poll - // thread rewrites slots while allocation threads may still read them under - // an acquire count load that observed the OLD count - plain accesses there - // are a C++ data race (UB), not just a staleness artifact. jint _watched_tids[KlassCandidate::MAX_QUALIFYING_TIDS]; volatile int _watched_tid_count; volatile bool _urgent_tracking; @@ -469,9 +471,23 @@ class alignas(alignof(SpinLock)) LivenessTracker { typedef struct KlassCountScratch { u32 klass_id; // Distinct GC ages (generations) of surviving tracked instances - // of this klass at this epoch. The size of this vector - // is the klass' generation count. - std::vector ages; + // of this klass at this epoch. The count (ages_count, saturated + // at MAX_DISTINCT_AGES) is the klass' generation count. + // + // Fixed-size (was a std::vector): push_back ran while the + // EXCLUSIVE table lock is held, including on the forced + // cleanup_table(true, false) path invoked synchronously from + // track()'s table-overflow branch on the JVMTI SampledObjectAlloc + // callback stack - a malloc there is exactly what the project's + // allocation-free GC-path preference forbids. The generation-count + // signal saturates: a klass with >= MAX_DISTINCT_AGES distinct + // surviving ages in a single epoch is already an extreme leak + // signal, and the population ring it feeds is only 30 samples + // (KLASS_POPULATION_RING_SIZE). + static constexpr int MAX_DISTINCT_AGES = 64; + u32 ages[MAX_DISTINCT_AGES]; + int ages_count; + bool ages_saturated; // set once ages_count hits MAX_DISTINCT_AGES // Top-N oldest surviving instances of this klass seen this epoch, // sorted by age descending. Used by foldKlassCountsLocked() to mint // representatives biased toward long-lived instances (Lindy effect: @@ -516,6 +532,26 @@ class alignas(alignof(SpinLock)) LivenessTracker { } KlassCountScratch; KlassCountScratch _klass_count_scratch[MAX_KLASS_POPULATION_ENTRIES]; int _klass_count_scratch_size; + // Open-addressed direct index over _klass_count_scratch, keyed by + // klass_id: slot value is scratch_index + 1 (0 = empty), hashed by the + // same mix() convention the frontier uses. accumulateKlassCount() runs + // once per SURVIVING table entry (up to MAX_TRACKING_TABLE_SIZE = 262144) + // while the exclusive table lock is held; the old linear scan over the + // 256 scratch entries made a fully-populated sweep O(entries x klasses). + // With the index each accumulate is one probe chain. Rebuilt by + // klassCountScratchReset(); MAX_KLASS_POPULATION_ENTRIES = 256 entries + // in a 512-slot table keeps the load factor at 0.5. + static constexpr int KLASS_COUNT_INDEX_SLOTS = MAX_KLASS_POPULATION_ENTRIES * 2; + u16 _klass_count_index[KLASS_COUNT_INDEX_SLOTS]; + + // Zeroes the scratch (size + every KlassCountScratch field the fold + // reads) and the direct index. Every _klass_count_scratch_size = 0 site + // must go through this so the index can never name a stale slot. + void klassCountScratchReset(); + + // Index lookup: returns the scratch slot for klass_id or nullptr (and + // takes an empty slot when `allocate` and scratch has capacity). + KlassCountScratch *klassCountScratchSlot(u32 klass_id, bool allocate); // Profiler::classMap()'s generation as of the last cleanup_table() call // that checked it, mirroring ReferenceChainTracker::_last_class_map_generation @@ -552,15 +588,15 @@ class alignas(alignof(SpinLock)) LivenessTracker { // instance of the same class) and correlate chains with HeapLiveObject. static constexpr int LEAK_TAG_POOL_SIZE = 256; static constexpr jlong LEAK_TAG_BASE = 0x40000000LL; - // Serializes the pool's free list and _leak_tag_info entries. NOT _table_lock: - // acquireLeakTag() runs under the SHARED table lock (tagLeakInstances) while - // releaseLeakTag() runs under the EXCLUSIVE one (cleanup_table) - a shared - // holder excludes the exclusive one, but only this dedicated lock makes the - // pool safe for any future second shared-lock mutator, and getLeakTagInfo() - // takes NO table lock at all (BFS poll thread). Lock order: _table_lock - // (any mode) is always acquired BEFORE _leak_tag_pool_lock, never the - // reverse; no path acquires _table_lock while holding the pool lock. - mutable SpinLock _leak_tag_pool_lock; + // Own lock for the pool state below (free list + info slots). The pool + // is mutated from cleanup_table()'s reaper pass (under the EXCLUSIVE + // table lock), tagLeakInstances()'s tagging state machine (deliberately + // run OUTSIDE the table lock - see its phase comment), and read by + // getLeakTagInfo() from ReferenceChainTracker's coverage path. A dedicated + // tiny spinlock keeps those contexts independent of _table_lock ordering + // (taking _table_lock inside/outside inconsistently would risk lock-order + // inversion); the critical sections are a few integer ops, no JNI. + SpinLock _leak_tag_pool_lock; int _leak_tag_free_list[LEAK_TAG_POOL_SIZE]; int _leak_tag_free_count; // Side table: for each tag in the pool, the (call_trace_id, tid) of @@ -600,7 +636,8 @@ class alignas(alignof(SpinLock)) LivenessTracker { // upcalls flush_table() already makes safely are just as safe there - see // that method's own comment for why a third caller needs both bypassing // the early-exit *and* resolution. - void cleanup_table(bool force = false, bool allow_resolve = true); + void cleanup_table(bool force = false, bool allow_resolve = true, + bool account_epoch = true); void flush_table(std::set *tracked_thread_ids); @@ -707,7 +744,7 @@ class alignas(alignof(SpinLock)) LivenessTracker { // --- Slope computation and candidate ranking (selectLeakCandidates() below) --- - // Per-tid sustained-trend gate half #1: the same regression growth + // Per-tid sustained-trend gate half #1: the same mean-of-thirds growth // test hasQualifyingGrowth() below applies to a klass's ring, at // KlassPopulationEntry::TidTrend granularity (TID_TREND_MIN_FILL_FOR_TREND // samples of that smaller ring, same LEAK_GROWTH_REL_MIN/ABS_MIN growth @@ -752,13 +789,12 @@ class alignas(alignof(SpinLock)) LivenessTracker { // The sustained-trend gate (this class's own header comment above, // "Sustained-trend gate") - both-required growth-magnitude and floor-rise - // tests, design doc's original "mean of thirds" choice since replaced by - // full-window least-squares regression (see ringThirdsStats, - // livenessTracker.cpp - cheap, allocation-free, one pass over the + // tests, design doc's explicit "mean of thirds" choice over full + // least-squares regression (cheap, allocation-free, one pass over the // ring, no sorting or extra storage). A single scan - // (ringThirdsStats(), livenessTracker.cpp) both derives the pass/fail - // result below AND updates entry.cached_slope (regression end value minus - // start value) for selectLeakCandidates()'s ranking, rather than + // (ringWindowStats(), livenessTracker.cpp) both derives the pass/fail + // result below AND updates entry.cached_slope (recent third's mean minus + // earliest third's mean) for selectLeakCandidates()'s ranking, rather than // that method re-scanning the same unchanged ring a moment later. Returns // false (leaving entry.cached_slope untouched) if entry.ring_fill is below // KLASS_POPULATION_MIN_FILL_FOR_TREND - not enough history yet to trust a @@ -921,9 +957,11 @@ class alignas(alignof(SpinLock)) LivenessTracker { // Look up the (call_trace_id, tid) recorded for a leak tag. Returns // false if the tag is not a valid leak tag or has been returned to the - // pool. Used by ReferenceChainTracker for coverage tracking. + // pool. Used by ReferenceChainTracker for coverage tracking. Takes the + // table lock in shared mode: the slot fields are written under the + // exclusive lock from several distinct threads (GC reaper paths). bool getLeakTagInfo(jlong tag, u64 *out_call_trace_id, - jint *out_tid) const; + jint *out_tid); // Reads _klass_population and writes up to `max` STABLE CLASS TAGS // (KlassPopulationEntry::stable_class_tag - NOT the classMap dictionary @@ -986,7 +1024,7 @@ class alignas(alignof(SpinLock)) LivenessTracker { // head/fill index, shared _heap_floor_time_ring timestamps - see // _container_mem_ring's own comment) so this compares _max_heap_bytes // against _container_memory_limit up front (the latter treated as - // unbounded when unavailable) and runs the regression-based rate + // unbounded when unavailable) and runs the "mean of thirds" rate // extrapolation (allocation-free, one ring scan) only once, against // whichever limit is smaller - not once per boundary. This matters // because container memory can grow from causes the heap-floor ring never @@ -1107,9 +1145,7 @@ class alignas(alignof(SpinLock)) LivenessTracker { return __atomic_load_n(&_watched_tid_count, __ATOMIC_ACQUIRE); } - jint watchedTidForTest(int i) const { - return __atomic_load_n(&_watched_tids[i], __ATOMIC_RELAXED); - } + jint watchedTidForTest(int i) const { return _watched_tids[i]; } static jlong leakTagBaseForTest() { return LEAK_TAG_BASE; } @@ -1346,7 +1382,7 @@ class alignas(alignof(SpinLock)) LivenessTracker { // reuses recordKlassPopulationSampleLocked()'s own creation branch // exactly; the seeded rising ramp that follows still clears // hasQualifyingGrowth() (a single 0 at the ring's start only lowers the - // regression start value, which RAISES the slope). + // earliest-third mean, which RAISES the slope). int slot = -1; for (int i = 0; i < _klass_population_size; i++) { if (_klass_population[i].klass_id == real_id) { @@ -1445,7 +1481,7 @@ class alignas(alignof(SpinLock)) LivenessTracker { void klassPopulationResetForTest() { _table_lock.lock(); _klass_population_size = 0; - _klass_count_scratch_size = 0; + klassCountScratchReset(); _test_klass_alias_count = 0; _table_lock.unlock(); // Also reset the heap-floor ring: it is a sibling piece of the same diff --git a/ddprof-lib/src/main/cpp/objectSampler.cpp b/ddprof-lib/src/main/cpp/objectSampler.cpp index 4232581200..3a11cc2fb5 100644 --- a/ddprof-lib/src/main/cpp/objectSampler.cpp +++ b/ddprof-lib/src/main/cpp/objectSampler.cpp @@ -182,13 +182,14 @@ Error ObjectSampler::start(Arguments &args) { return error; } if (_interval > 0) { - if (_record_liveness || _gc_generations) { - error = LivenessTracker::instance()->start(args); - if (error) { - return error; - } - } - + // Always call through, even when this start's own args request neither + // liveness recording nor gc generations: LivenessTracker::start() -> + // initialize() refreshes its own _gc_generations/_enabled from args + // unconditionally (see that method's own comment) and is a no-op beyond + // that when disabled. Gating this call on ObjectSampler's own + // (freshly-set, correct) flags left LivenessTracker's flags stuck at + // whatever the previous recording in this process last set them to, + // since it never got a chance to observe this recording's request at all. jvmtiEnv *jvmti = VM::jvmti(); // JVMTI Object Sampler is a 'solo' feature, meaning that it can only be // used by one JVMTI environment. Therefore, we can rely on the fact that if @@ -198,9 +199,18 @@ Error ObjectSampler::start(Arguments &args) { JVMTI_EVENT_SAMPLED_OBJECT_ALLOC, NULL); __atomic_store_n(&_active, true, __ATOMIC_RELEASE); __atomic_store_n(&_last_config_update_ts, OS::nanotime(), __ATOMIC_RELEASE); + // Started LAST, after every step that can fail above: the old order + // (tracker first) left LivenessTracker started-but-never-driven if the + // JVMTI enabling failed - its GC-callback machinery and table would run + // with no sampler feeding it until the next stop(). stop() still stops + // it unconditionally (LivenessTracker::stop() self-guards on _enabled). // need to reset the running sum in order for 'updateConfiguration' to be // able to generate proper diffs _alloc_event_count = 0; + error = LivenessTracker::instance()->start(args); + if (error) { + return error; + } } return Error::OK; @@ -212,9 +222,9 @@ void ObjectSampler::stop() { jvmti->SetEventNotificationMode(JVMTI_DISABLE, JVMTI_EVENT_SAMPLED_OBJECT_ALLOC, NULL); - if (_record_liveness || _gc_generations) { - LivenessTracker::instance()->stop(); - } + // See start()'s own comment on why this call is unconditional - + // LivenessTracker::stop() already self-guards on its own _enabled. + LivenessTracker::instance()->stop(); } Error ObjectSampler::updateConfiguration(u64 events, double time_coefficient) { diff --git a/ddprof-lib/src/main/cpp/os_linux.cpp b/ddprof-lib/src/main/cpp/os_linux.cpp index daabc8f5b7..db56ed8a06 100644 --- a/ddprof-lib/src/main/cpp/os_linux.cpp +++ b/ddprof-lib/src/main/cpp/os_linux.cpp @@ -35,6 +35,7 @@ #include "guards.h" #include "log.h" #include "os.h" +#include "spinLock.h" #ifndef __musl__ #include @@ -916,15 +917,23 @@ int OS::getCgroupCpuMillicores() { // cgroups whose usage the leaf's memory.current excludes - pairing the // ancestor limit with leaf usage would overstate the available memory and // delay the OOM projection. +// +// Both globals below are written by getContainerMemoryLimit() (called from +// LivenessTracker::start() on the profiler-start thread and from +// SanityCheckTest's setup) and read by getContainerMemoryUsage() (from the +// JVMTI GarbageCollectionFinish callback, on whatever thread triggered +// GC) - genuinely concurrent threads, so every access is serialized by +// g_container_mem_lock. Without it the reader could observe a torn, +// half-overwritten path string. The usage-leaf cache additionally keeps +// the per-GC cost at one open/read/close: getContainerMemoryUsage() reads +// the remembered leaf directly instead of re-deriving it via +// getOwnCgroupPath() (/proc/self/cgroup parsing) on every GC. +static SpinLock g_container_mem_lock; static char g_memory_limit_cgroup_path[PATH_MAX] = {0}; - -// Which cgroup hierarchy (v2 or v1) supplied the winning limit recorded in -// g_memory_limit_cgroup_path: the usage file must be read from the SAME -// hierarchy (v2 memory.current vs v1 memory.usage_in_bytes), not probed by -// filename order - on a hybrid system both controller files can be visible -// for the same cgroup dir, and reading the wrong one pairs the limit with an -// unrelated usage number. -static bool g_memory_limit_cgroup_v2 = true; +// The leaf whose memory.current/memory.usage_in_bytes pairs with the +// winning limit (the winner cgroup itself, or - when unconstrained - the +// process's own leaf). Empty until the first successful limit walk. +static char g_usage_leaf_path[PATH_MAX] = {0}; static long walkCgroupV2MemoryLimit(char* path, char* winner_path_out) { size_t base_len = strlen("/sys/fs/cgroup"); @@ -991,11 +1000,16 @@ static long walkCgroupV1MemoryLimit(char* path, char* winner_path_out) { long OS::getContainerMemoryLimit() { char subpath[PATH_MAX]; char path[PATH_MAX]; + // Local winner/usage-leaf bookkeeping, published under the lock at the + // end - never write the shared globals directly (the usage reader runs + // concurrently on GC-callback threads). + char winner[PATH_MAX] = {0}; + char usage_leaf[PATH_MAX] = {0}; + long result; // Recomputed on every call; getContainerMemoryUsage() pairs its usage // read with whatever path won here (see the winner-path comment on // walkCgroupV2MemoryLimit()). - g_memory_limit_cgroup_path[0] = '\0'; // Try cgroup v2 first, resolved from this process's own cgroup path. if (getOwnCgroupPath("", subpath, sizeof(subpath))) { @@ -1010,8 +1024,12 @@ long OS::getContainerMemoryLimit() { int fd = open(leaf, O_RDONLY); if (fd != -1) { close(fd); - g_memory_limit_cgroup_v2 = true; - return walkCgroupV2MemoryLimit(path, g_memory_limit_cgroup_path); + result = walkCgroupV2MemoryLimit(path, winner); + // The usage read pairs with the winner cgroup when the + // walk found a limit; otherwise the process's own leaf. + snprintf(usage_leaf, sizeof(usage_leaf), "%s", + winner[0] != '\0' ? winner : path); + goto publish; } } } @@ -1031,14 +1049,27 @@ long OS::getContainerMemoryLimit() { int fd = open(leaf, O_RDONLY); if (fd != -1) { close(fd); - g_memory_limit_cgroup_v2 = false; - return walkCgroupV1MemoryLimit(path, g_memory_limit_cgroup_path); + result = walkCgroupV1MemoryLimit(path, winner); + snprintf(usage_leaf, sizeof(usage_leaf), "%s", + winner[0] != '\0' ? winner : path); + goto publish; } } } } - return -1; + // Unconstrained or unavailable - remember that (empty paths) so the + // usage reader skips straight to its own leaf re-derivation. + result = -1; + winner[0] = '\0'; + usage_leaf[0] = '\0'; + +publish: + g_container_mem_lock.lock(); + memcpy(g_memory_limit_cgroup_path, winner, sizeof(winner)); + memcpy(g_usage_leaf_path, usage_leaf, sizeof(usage_leaf)); + g_container_mem_lock.unlock(); + return result; } // Reads the current usage from the same cgroup level that supplied @@ -1053,26 +1084,42 @@ long OS::getContainerMemoryUsage() { char subpath[PATH_MAX]; char path[PATH_MAX]; + // Fast path: the leaf remembered by the last limit walk (winner cgroup, + // or the process's own leaf when unconstrained). Snapshot under the + // lock - the walk may be republishing concurrently, and an unsynchronized + // read could observe a torn half-written path. This keeps the per-GC + // cost at one open/read/close instead of re-deriving the cgroup path + // via getOwnCgroupPath() (/proc/self/cgroup parsing) on every GC. + char leaf[PATH_MAX] = {0}; + g_container_mem_lock.lock(); + if (g_usage_leaf_path[0] != '\0') { + memcpy(leaf, g_usage_leaf_path, sizeof(leaf)); + } else if (g_memory_limit_cgroup_path[0] != '\0') { + memcpy(leaf, g_memory_limit_cgroup_path, sizeof(leaf)); + } + g_container_mem_lock.unlock(); + // Same cgroup the winning limit came from, if the limit walk recorded // one - read its usage first, falling back to the process's own leaf. - // The usage filename follows the hierarchy that supplied the limit (the - // recorded winner's hierarchy is authoritative, not filename order). - if (g_memory_limit_cgroup_path[0] != '\0') { + // v2 names the file memory.current, v1 memory.usage_in_bytes - try both. + if (leaf[0] != '\0') { char file[PATH_MAX]; - const char *fmt = g_memory_limit_cgroup_v2 ? "%s/memory.current" - : "%s/memory.usage_in_bytes"; - if ((size_t)snprintf(file, sizeof(file), fmt, - g_memory_limit_cgroup_path) < sizeof(file)) { + const char *usage_files[] = {"%s/memory.current", "%s/memory.usage_in_bytes"}; + for (const char *fmt : usage_files) { + if ((size_t)snprintf(file, sizeof(file), fmt, leaf) >= sizeof(file)) { + continue; + } int fd = open(file, O_RDONLY); - if (fd != -1) { - char buf[32] = {0}; - ssize_t r = read(fd, buf, sizeof(buf) - 1); - close(fd); - if (r > 0) { - long usage = atol(buf); - if (usage >= 0) { - return usage; - } + if (fd == -1) { + continue; + } + char buf[32] = {0}; + ssize_t r = read(fd, buf, sizeof(buf) - 1); + close(fd); + if (r > 0) { + long usage = atol(buf); + if (usage >= 0) { + return usage; } } } diff --git a/ddprof-lib/src/main/cpp/os_macos.cpp b/ddprof-lib/src/main/cpp/os_macos.cpp index 3d3aa8ef6b..575acd805b 100644 --- a/ddprof-lib/src/main/cpp/os_macos.cpp +++ b/ddprof-lib/src/main/cpp/os_macos.cpp @@ -385,7 +385,12 @@ long OS::getContainerMemoryLimit() { } long OS::getContainerMemoryUsage() { - return -1; // macOS has no cgroup support. + // Contract mirror of the Linux implementation: "no boundary available" + // is -1 (treated by LivenessTracker::secondsToOOM() as no projection, + // never as unbounded), not 0 - a 0 here would read as a container at + // zero usage and skew the container-boundary ring's slope toward + // noise. macOS has no cgroup support. + return -1; } u64 OS::getProcessCpuTime(u64* utime, u64* stime) { diff --git a/ddprof-lib/src/main/cpp/profiler.cpp b/ddprof-lib/src/main/cpp/profiler.cpp index 3d1472a644..efae653a26 100644 --- a/ddprof-lib/src/main/cpp/profiler.cpp +++ b/ddprof-lib/src/main/cpp/profiler.cpp @@ -99,6 +99,14 @@ void Profiler::onThreadStart(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { updateThreadName(jvmti, jni, thread, true); } + // Registers the tid -> Thread-object global ref that the reference-chain + // engine's walkCandidateThreadLocals() descends from for + // candidate-scoped ThreadLocalMap reach. No-op while reference chains are + // disabled (checked inside the tracker); jni/thread may be null on the + // internal pre-existing-threads call from start(), which the tracker + // also refuses. + ReferenceChainTracker::instance()->registerThreadObject(jni, tid, thread); + _cpu_engine->registerThread(tid); _wall_engine->registerThread(tid); } @@ -112,6 +120,11 @@ void Profiler::onThreadEnd(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { // ProfiledThread is alive - do full cleanup and use efficient tid access int slot_id = current->filterSlotId(); tid = current->tid(); + // NOT gated on reference-chains enabled: a thread registered while a + // recording ran must release its global ref when it ends, even if the + // recording has since stopped (see unregisterThreadObject()'s comment, + // referenceChains.h). + ReferenceChainTracker::instance()->unregisterThreadObject(jni, tid); if (_thread_filter.enabled()) { _thread_filter.unregisterThread(slot_id); @@ -139,6 +152,11 @@ void Profiler::onThreadEnd(jvmtiEnv *jvmti, JNIEnv *jni, jthread thread) { return; } + // Same rationale as the ProfiledThread-alive branch above: a thread + // registered during an active recording must release its global ref when + // it ends, whatever path its teardown takes. + ReferenceChainTracker::instance()->unregisterThreadObject(jni, tid); + updateThreadName(jvmti, jni, thread, false); _cpu_engine->unregisterThread(tid); _wall_engine->unregisterThread(tid); @@ -869,6 +887,117 @@ void Profiler::writeHeapUsage(long value, bool live) { _locks[lock_index].unlock(); } +void Profiler::writeReferenceChainAbandoned(ReferenceChainAbandonedEvent *event) { + int tid = ProfiledThread::currentTid(); + if (tid < 0) { + return; + } + // Same bounded-retry pattern as writeReferenceChain() below: the caller + // (Profiler::dump()'s drain loop, referenceChains.cpp + // drainPendingAbandonedEvents()) has already removed the event from the + // pending queue, so a bare non-blocking 3-slot sweep that misses all + // three locks would lose it permanently - unlike resolved-chain events + // (which snapshot-and-keep), an abandoned event has no second chance. + u32 lock_index; + bool locked = false; + int sweeps = 0; + u64 start_ns = OS::nanotime(); + // Per-event budget: abandoned events are rare (one per abandoned search) + // and carry no batch deadline from the caller. + const u64 kAbandonedWriteBudgetNs = 50 * 1000000ULL; + u64 deadline_ns = OS::nanotime() + kAbandonedWriteBudgetNs; + for (;;) { + sweeps++; + lock_index = getLockIndex(tid); + if (_locks[lock_index].tryLock() || + _locks[lock_index = (lock_index + 1) % CONCURRENCY_LEVEL].tryLock() || + _locks[lock_index = (lock_index + 2) % CONCURRENCY_LEVEL].tryLock()) { + locked = true; + break; + } + if (OS::nanotime() >= deadline_ns) { + break; + } + usleep(1000); + } + if (!locked) { + Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); + TEST_LOG("Profiler::writeReferenceChainAbandoned drop: lock contention " + "exhausted budget after sweeps=%d waited_us=%llu", + sweeps, (unsigned long long)((OS::nanotime() - start_ns) / 1000)); + return; + } + _jfr.recordReferenceChainAbandoned(lock_index, event); + _locks[lock_index].unlock(); +} + +// Unlike writeReferenceChainAbandoned() above (mirroring CPU/wall's signal-handler-safe +// non-blocking pattern out of caution, even though its own call site - Profiler::dump(), +// profiler.cpp - isn't a signal handler either), this call site genuinely cannot be one: +// this is called from Profiler::dump()'s drain loop, on dump()'s own calling thread, once +// per event snapshotted from ReferenceChainTracker::_resolved_chains (up to +// MAX_RESOLVED_CHAINS per dump) - never from pollWatchedTargets() or any other call on +// ReferenceChainTracker's own BFS agent thread, and never from a signal handler. A single +// bare 3-slot tryLock() sweep with no wait - correct for a signal handler, which must never +// block - was found, by running the end-to-end integration test for real, +// to drop this event under perfectly ordinary contention: the same _locks[] pool is shared +// with every other sample type (recordJVMTISample() et al.), and any nontrivial allocation +// throughput keeps enough of CONCURRENCY_LEVEL's slots busy that 3 immediate, back-to-back +// attempts routinely all miss. A bounded retry with a short sleep between sweeps costs +// nothing the dump()-thread cannot afford, but the retry budget below is a single deadline +// shared across the *entire* drain batch (see the caller in dump()) rather than per event: +// with up to MAX_RESOLVED_CHAINS events snapshotted, a fresh per-event budget could stall +// the dump/JFR-flush thread for seconds under contention. Once the shared deadline has +// passed this degrades to the same single non-blocking 3-slot sweep as +// writeReferenceChainAbandoned() above for the remainder of the batch. +void Profiler::writeReferenceChain(ReferenceChainEvent *event, u64 deadline_ns) { + int tid = ProfiledThread::currentTid(); + if (tid < 0) { + TEST_LOG("Profiler::writeReferenceChain drop: currentTid() < 0"); + return; + } + u32 lock_index; + bool locked = false; + int sweeps = 0; + u64 start_ns = OS::nanotime(); + for (;;) { + sweeps++; + lock_index = getLockIndex(tid); + if (_locks[lock_index].tryLock() || + _locks[lock_index = (lock_index + 1) % CONCURRENCY_LEVEL].tryLock() || + _locks[lock_index = (lock_index + 2) % CONCURRENCY_LEVEL].tryLock()) { + locked = true; + break; + } + if (OS::nanotime() >= deadline_ns) { + // Shared batch budget exhausted - the sweep just above was already a + // single non-blocking attempt, so stop retrying rather than sleeping + // again. + break; + } + usleep(1000); + } + if (!locked) { + // Unlike the drain-once era, this drop is NOT permanent: the event was + // only copied out of ReferenceChainTracker::_resolved_chains + // (drainPendingChainEvents() snapshots without clearing), so as long as + // the sample stays live the next dump re-emits it and gets another chance + // at the lock. Still counted like every other counted-drop path + // (REFERENCE_CHAIN_WRITE_DROPPED's own comment) rather than dropping it + // silently. + Counters::increment(REFERENCE_CHAIN_WRITE_DROPPED); + TEST_LOG("Profiler::writeReferenceChain drop: lock contention exhausted shared " + "deadline after sweeps=%d waited_us=%llu", + sweeps, (unsigned long long)((OS::nanotime() - start_ns) / 1000)); + return; + } + TEST_LOG("Profiler::writeReferenceChain locked lock_index=%u after sweeps=%d " + "waited_us=%llu", + lock_index, sweeps, (unsigned long long)((OS::nanotime() - start_ns) / 1000)); + _jfr.recordReferenceChain(lock_index, event); + _locks[lock_index].unlock(); +} + bool Profiler::prewarmUnwinder() { #ifdef __linux__ // Force libgcc_s.so.1 to load now and report whether that succeeded. This @@ -1703,26 +1832,6 @@ Error Profiler::start(Arguments &args, bool reset) { } } } - if ((activated & EM_ALLOC) && args._reference_chains) { - // Reference-chain tracking chases LivenessTracker's leak-tagged - // candidates, so it only runs when allocation sampling actually - // activated. Must run AFTER ObjectSampler::start() -> - // LivenessTracker::start() (ordering noted in referenceChains.cpp) - - // the block above is that ordering point. - error = ReferenceChainTracker::instance()->start(args); - if (error) { - Log::warn("%s", error.message()); - error = Error::OK; // recoverable - recording continues without chains - } else { - ReferenceChainTracker::instance()->startThread(); - // Pre-existing threads must be registered from the profiler lifecycle - // (registerExistingThreads()'s own comment): Profiler::onThreadStart() - // only sees threads started after the recording began. - ReferenceChainTracker::instance()->registerExistingThreads(VM::jvmti(), - VM::jni()); - _reference_chains_active = true; - } - } if (_event_mask & EM_NATIVEMEM) { error = malloc_tracer.start(args); if (error) { @@ -1770,6 +1879,42 @@ Error Profiler::start(Arguments &args, bool reset) { // Paired with drainInflight() on the stop side. _cpu_engine->enableEvents(true); + // Independent of the CPU/wall/alloc engine mask above (GC-triggered, not + // sample-triggered) - same pattern as malloc_tracer/NativeSocketSampler + // being gated on their own flags rather than folded into `activated`. + // Placed after the engines are confirmed running (inside this + // `if (activated)` block) so there is nothing to unwind here if it + // fails - see this method's failure path below, which never reaches + // this point. + // Called unconditionally, not gated on args._reference_chains: start() + // is the only place that refreshes ReferenceChainTracker::_enabled + // (stop() deliberately leaves it unchanged - see that method's own + // comment), so a previous recording's `referencechains=true` session + // must still reach start() when this one opts out, or the tracker keeps + // reporting enabled()==true - and Profiler::dump()'s reference-chains + // gate keeps emitting that stale session's cached chains/abandonment + // state - for the entire duration of this new, opted-out recording. + // start() itself sets `_enabled = args._reference_chains` up front and + // returns early when that is false, so this call is a cheap no-op for + // an opted-out recording. + error = ReferenceChainTracker::instance()->start(args); + if (error) { + Log::warn("%s", error.message()); + error = Error::OK; // recoverable + } else if (args._reference_chains) { + // Only safe once the JVM/JVMTI environment is fully up, which is + // guaranteed at this point in Profiler::start() - see + // ReferenceChainTracker::start()'s own comment (referenceChains.cpp) + // for why this is not called from inside start() itself. + ReferenceChainTracker::instance()->startThread(); + // Pre-existing threads (alive since before this recording began) + // never fired onThreadStart() - same lifecycle rationale as + // startThread() above for why this runs here rather than inside + // ReferenceChainTracker::start(). + ReferenceChainTracker::instance()->registerExistingThreads( + VM::jvmti(), VM::jni()); + } + _state.store(RUNNING, std::memory_order_release); _start_time = time(NULL); __atomic_add_fetch(&_epoch, 1, __ATOMIC_RELAXED); @@ -1812,17 +1957,48 @@ Error Profiler::stop() { if (_event_mask & EM_ALLOC) _alloc_engine->stop(); - if (_reference_chains_active) { - // Join the BFS thread and clear the recording-boundary state before the - // rest of the teardown (stopThread() wakes and joins; stop() resets the - // per-recording caches). Matches the Profiler::stop() order documented - // in referenceChains.cpp's stopThread()/stop() comments. + if (_event_mask & EM_NATIVEMEM) + malloc_tracer.stop(); + // Not part of _event_mask (see the matching start() block above) - gated + // on enabled() instead, which start() set from args._reference_chains for + // this session. + if (ReferenceChainTracker::instance()->enabled()) { ReferenceChainTracker::instance()->stopThread(); ReferenceChainTracker::instance()->stop(); - _reference_chains_active = false; + // Drains the global refs of threads that ended during this recording + // (referenceChains.h, _thread_refs_pending_delete). Safe here: the BFS + // thread was joined by stopThread() above, so no walk phase can still + // hold a copied Thread-object ref. + ReferenceChainTracker::instance()->releaseEndedThreadRefs(VM::jni()); + // Final drain: only dump() writes the tracker's resolved-chain cache and + // abandoned-event queue, so a recording that ends without a preceding + // dump() would lose every reference-chain result discovered since the + // last dump. Both writes go through the same JFR write paths dump() + // uses, before the final chunk is finalized by _jfr.stop() below. The + // tracker is fully stopped here, so the caches are stable snapshots. + std::vector pending_abandoned_events; + ReferenceChainTracker::instance()->drainPendingAbandonedEvents( + &pending_abandoned_events); + for (auto &rc_event : pending_abandoned_events) { + writeReferenceChainAbandoned(&rc_event); + } + std::vector pending_chain_events; + ReferenceChainTracker::instance()->drainPendingChainEvents( + &pending_chain_events); + const u64 kChainDrainBudgetNs = 50 * 1000000ULL; + u64 chain_drain_deadline_ns = OS::nanotime() + kChainDrainBudgetNs; + for (auto &rc_event : pending_chain_events) { + writeReferenceChain(&rc_event, chain_drain_deadline_ns); + } + // Threads still alive at stop keep registered global refs that nothing + // else will release: the tracker's BFS thread is gone (no walks need + // them) and threads that end AFTER this point take the fallback + // onThreadEnd path, whose unregister can no longer be relied on for + // entries a future recording did not re-register. Delete every + // remaining registry entry so ended threads cannot stay reachable + // anchors for the rest of the JVM's life. + ReferenceChainTracker::instance()->releaseAllThreadObjects(VM::jni()); } - if (_event_mask & EM_NATIVEMEM) - malloc_tracer.stop(); // Stop the refresher BEFORE socket unpatch: the refresher calls // install_socket_hooks() which re-reads _socket_active before acquiring the // patch lock. If the refresher runs concurrently with unpatch_socket_functions() @@ -1982,29 +2158,60 @@ Error Profiler::dump(const char *path, const int length) { // by the live objects LivenessTracker::instance()->flush(thread_ids); - // Emit the reference-chain tracker's pending events into this dumping - // chunk: chain events are snapshot-and-kept (re-emitted into every chunk - // while the sample stays live), abandonment events are a true drain. - // Runs before rotateDictsAndRun() so the events land inside the chunk - // being written, and under a profiler lock like every other - // recording-buffer writer (dump runs on a normal thread holding only - // _state_lock; the _state_lock -> _locks order is the codebase's). - { - int dump_tid = ProfiledThread::currentTid(); - u32 lock_index = getLockIndex(dump_tid >= 0 ? dump_tid : 0); - _locks[lock_index].lock(); - std::vector chain_events; - ReferenceChainTracker::instance()->drainPendingChainEvents(&chain_events); - for (auto &event : chain_events) { - _jfr.recordReferenceChain(lock_index, &event); - } - std::vector abandoned_events; + // ReferenceChainTracker::_resolved_chains (and the search-state fields + // read below) are intentionally left populated across a stop()/start() + // cycle - see _resolved_chains' own comment (referenceChains.h) - but + // that means they can still hold state from a *previous* recording that + // had referencechains enabled, even once the current recording started + // with referencechains=false (in which case ReferenceChainTracker:: + // start() sets _enabled=false and no BFS thread is polling to ever + // refresh or prune them). Gate both emissions on the current session's + // flag so an opted-out recording does not keep re-reporting a dead + // session's abandoned search or stale resolved chains. + if (ReferenceChainTracker::instance()->enabled()) { + // ReferenceChainTracker's BFS thread restarts an ABANDONED search on + // its own ~1s cadence (referenceChains.cpp shouldRunPass() -> + // restartSearch()), which clears the very state + // buildAbandonedEvent() needs. A live re-read of searchState() here + // would almost always miss that ~1s window against dump()'s much + // slower JFR-chunk-rotation cadence. Instead each abandon is + // snapshotted into a queue at the moment it happens + // (enqueuePendingAbandonedEvent(), called from runPass()) and drained + // here - a true drain, unlike drainPendingChainEvents() below, since + // an abandon is a one-off past occurrence rather than an ongoing live + // sample. + std::vector pending_abandoned_events; ReferenceChainTracker::instance()->drainPendingAbandonedEvents( - &abandoned_events); - for (auto &event : abandoned_events) { - _jfr.recordReferenceChainAbandoned(lock_index, &event); + &pending_abandoned_events); + for (auto &rc_event : pending_abandoned_events) { + // No re-stamp here: the event's _start_time was set when the search + // actually stopped (enqueuePendingAbandonedEvent()) - re-stamping at + // dump time would misreport a seconds-old abandon as happening now. + writeReferenceChainAbandoned(&rc_event); } - _locks[lock_index].unlock(); + + // Re-emit every currently-cached datadog.ReferenceChain pollWatchedTargets() + // (referenceChains.cpp) has resolved - snapshotted here, on this call's + // own thread, rather than written eagerly from the BFS scheduling thread + // that discovered them (see ReferenceChainTracker::_resolved_chains' own + // comment for why the cache re-emits on every dump rather than draining). + std::vector pending_chain_events; + ReferenceChainTracker::instance()->drainPendingChainEvents( + &pending_chain_events); + // One ~50ms retry budget for the *whole* batch, not per event - + // writeReferenceChain()'s own comment for why: up to + // MAX_RESOLVED_CHAINS events can be snapshotted, and a fresh per-event + // budget would let this dump()-thread stall for seconds under ordinary + // _locks[] contention. + const u64 kChainDrainBudgetNs = 50 * 1000000ULL; + u64 chain_drain_deadline_ns = OS::nanotime() + kChainDrainBudgetNs; + long long write_dropped_before = Counters::getCounter(REFERENCE_CHAIN_WRITE_DROPPED); + for (auto &rc_event : pending_chain_events) { + writeReferenceChain(&rc_event, chain_drain_deadline_ns); + } + TEST_LOG("Profiler::dump reference-chain batch=%d write_dropped=%lld", + (int)pending_chain_events.size(), + Counters::getCounter(REFERENCE_CHAIN_WRITE_DROPPED) - write_dropped_before); } Libraries::instance()->refresh(); @@ -2022,6 +2229,15 @@ Error Profiler::dump(const char *path, const int length) { err = _jfr.dump(path, length); __atomic_add_fetch(&_epoch, 1, __ATOMIC_SEQ_CST); }); + if (err) { + // Log::debug, not just TEST_LOG: TEST_LOG is compiled out of release + // builds entirely, so a JFR dump failure - the event stream this + // whole subsystem exists to produce - would fail silently in + // production. debug-level logging keeps it out of the steady-state + // log while still being reachable when someone turns debug on. + Log::debug("Profiler::dump _jfr.dump failed: %s", err.message()); + TEST_LOG("Profiler::dump _jfr.dump failed: %s", err.message()); + } _thread_info.clearAll(thread_ids); _thread_info.reportCounters(); diff --git a/ddprof-lib/src/main/cpp/profiler.h b/ddprof-lib/src/main/cpp/profiler.h index c679009e79..965806bf14 100644 --- a/ddprof-lib/src/main/cpp/profiler.h +++ b/ddprof-lib/src/main/cpp/profiler.h @@ -195,7 +195,9 @@ class alignas(alignof(SpinLock)) Profiler { // // rotate() is self-contained: it uses _accepting + RefCountGuard to drain // concurrent JNI readers, and SignalBlocker prevents profiling signals on - // this thread from inserting into old_active between Phase 1 and Phase 2. + // this thread from inserting into old_active between the pre-populate copy + // step and the catch-up copy step of the dictionary's two-step rotation + // (see StringDictionary::rotate(), stringDictionary.h). // No external lock is required for rotation. // // lockAll() wraps jfr_op only — to gate call-trace writers (signal handlers @@ -470,6 +472,20 @@ class alignas(alignof(SpinLock)) Profiler { void writeDatadogProfilerSetting(int tid, int length, const char *name, const char *value, const char *unit); void writeHeapUsage(long value, bool live); + // Mirrors writeHeapUsage()'s shape exactly. Called from dump() whenever + // ReferenceChainTracker's search has ended in SearchState::ABANDONED, + // the same way LivenessTracker::flush() is called from dump(). + void writeReferenceChainAbandoned(ReferenceChainAbandonedEvent *event); + // Unlike writeReferenceChainAbandoned() above, this is NOT a bare 3-slot + // tryLock() sweep - it retries with a bounded, sleeping loop because its + // call site is dump()'s drain loop (profiler.cpp), on dump()'s own calling + // thread, which can tolerate blocking, unlike a signal handler; see this + // method's own comment in profiler.cpp for why that retry exists. + // `deadline_ns` is a single retry budget shared across dump()'s *entire* + // drain batch (not reset per event) - see the caller in dump() and this + // method's own comment in profiler.cpp for why a per-event budget would be + // unbounded across a large batch. + void writeReferenceChain(ReferenceChainEvent *event, u64 deadline_ns); int eventMask() const { return _event_mask; } bool isRemoteSymbolication() const { return _remote_symbolication; } bool sanityCheckFailed() const { return _sanity_check_failed; } diff --git a/ddprof-lib/src/main/cpp/referenceChainAnchors.cpp b/ddprof-lib/src/main/cpp/referenceChainAnchors.cpp deleted file mode 100644 index 16dd391bf5..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainAnchors.cpp +++ /dev/null @@ -1,887 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -#include "referenceChains.h" -#include "referenceChainInternal.h" -#include "common.h" -#include "counters.h" -#include "jniHelper.h" -#include "jvmThread.h" -#include "livenessTracker.h" -#include "log.h" -#include "objectSampler.h" -#include "os.h" -#include "profiler.h" -#include "rcDebugLevel.h" -#include "tsc.h" -#include "vmEntry.h" -#include -#include -#include -#include -#include -#include -#include -#include - -// Candidate-scoped reach: bounded descend walks from anchor objects (see -namespace { - -// Class tags to neither admit nor descend into during a descend walk (see -// ReferenceChainPassContext::_no_descend_class_tags' own comment). -const char *const kNoDescendClassNames[] = { - "java/lang/ClassLoader", "java/lang/ThreadGroup", - "java/security/ProtectionDomain", -}; - -int resolveNoDescendClassTags(jvmtiEnv *jvmti, JNIEnv *jni, - jlong *out, int cap) { - int count = 0; - for (const char *name : kNoDescendClassNames) { - if (count >= cap) { - break; - } - jclass cls = jni->FindClass(name); - if (cls == nullptr) { - // Not loadable in this JVM (e.g. java.security classes stripped by a minimal runtime) - skip; - // the gate simply does not cover it. - jni->ExceptionClear(); - continue; - } - jlong tag = 0; - if (jvmti->GetTag(cls, &tag) == JVMTI_ERROR_NONE && tag != 0) { - out[count++] = tag; - } - jni->DeleteLocalRef(cls); - } - return count; -} - -// java.lang.ThreadLocal$ThreadLocalMap's class tag for walkCandidateThreadLocals()'s anchor gate -// (see ReferenceChainPassContext:: _anchor_descend_class_tag's own comment): the value type of BOTH -// of Thread's threadLocals and inheritableThreadLocals fields, and its exact class tag is what the -// anchor gate compares against. -jlong resolveThreadLocalMapClassTag(jvmtiEnv *jvmti, JNIEnv *jni) { - jlong tag = 0; - jclass cls = jni->FindClass("java/lang/ThreadLocal$ThreadLocalMap"); - if (cls == nullptr) { - jni->ExceptionClear(); - return 0; - } - jvmti->GetTag(cls, &tag); - jni->DeleteLocalRef(cls); - return tag; -} - -} // namespace - -void ReferenceChainTracker::descendFromAnchor( - jvmtiEnv *jvmti, JNIEnv *jni, jobject anchor, jlong anchor_tag, - u32 anchor_depth, jlong anchor_descend_class_tag, int budget, - int *edges_admitted, bool *truncated, bool *frontier_cap_hit, - u64 *safepoint_ticks) { - ReferenceChainPassContext ctx; - ctx.tracker = this; - ctx.frontier = _frontier; - // Bound admission to DESCENT_HOPS below the anchor, still subject to the global hop cap. - int descent_cap = (int)anchor_depth + DESCENT_HOPS; - ctx.hop_cap = descent_cap < _hop_cap ? descent_cap : _hop_cap; - ctx.budget = budget; - ctx.edges_admitted = 0; - ctx.truncated = false; - ctx.frontier_cap_hit = false; - - ctx._no_descend_class_tag_count = - resolveNoDescendClassTags(jvmti, jni, ctx._no_descend_class_tags, - ReferenceChainPassContext::NO_DESCEND_CLASS_CAP); - if (anchor_descend_class_tag != 0) { - ctx._descent_anchor_tag = anchor_tag; - ctx._anchor_descend_class_tag = anchor_descend_class_tag; - } - - jvmtiHeapCallbacks callbacks; - memset(&callbacks, 0, sizeof(callbacks)); - callbacks.heap_reference_callback = heapReferenceCallback; - u64 follow_start_ticks = TSC::ticks(); - jvmti->FollowReferences(0, nullptr, anchor, &callbacks, &ctx); - *safepoint_ticks += TSC::ticks() - follow_start_ticks; - *edges_admitted += ctx.edges_admitted; - *truncated = *truncated || ctx.truncated; - *frontier_cap_hit = *frontier_cap_hit || ctx.frontier_cap_hit; -} - -std::vector -ReferenceChainTracker::collectStaticFieldAnchorsForRotation(int max_count) { - std::vector selected; - if (max_count <= 0 || _static_anchor_index.empty()) { - return selected; - } - // Tiered selection over _static_anchor_index (O(anchors) per pass, under ONE shared lock - the - // lookups below are lookupLocked()). - size_t idx_size = _static_anchor_index.size(); - if (_anchor_container_cursor >= idx_size) { - _anchor_container_cursor = 0; - } - if (_anchor_other_cursor >= idx_size) { - _anchor_other_cursor = 0; - } - struct TierPick { - size_t pos; - jlong tag; - }; - std::vector leak_picks; - std::vector fresh_picks; - std::vector container_picks; - std::vector other_picks; - leak_picks.reserve(16); - // Fresh picks kept by the queue drain (bounded by max_count) - used to keep the fair-tier - // consumption below from double-selecting them. - std::unordered_set fresh_kept_tags; - const size_t fresh_queue_len = _static_anchor_fresh_queue.size(); - _frontier->withSharedLock([&](const FrontierTable *frontier) { - // Index scan: partition every eligible anchor into the leak tier or one of the two fair tiers - // (the fresh lane is decided by the queue drain below - a fresh-kept anchor also lands in a - // fair pick vector here and is skipped at consumption time via fresh_kept_tags). - for (size_t i = 0; i < idx_size; i++) { - jlong tag = _static_anchor_index[i]; - FrontierEntry entry{}; - if (!frontier->lookupLocked(tag, &entry) || - entry.parent_tag != 0 || - (entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && - entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) || - (entry.state != FrontierEntryState::FRONTIER && - entry.state != FrontierEntryState::EXPANDED) || - isQueuedForRotation(tag)) { - continue; - } - if (entry.leak_tag != 0) { - leak_picks.push_back(TierPick{i, tag}); - } else if (i < _static_anchor_own_class_tags.size()) { - auto shape_it = - _class_shape_cache.find(_static_anchor_own_class_tags[i]); - if (shape_it != _class_shape_cache.end() && - shape_it->second == (u8)AnchorClassShape::CONTAINER) { - container_picks.push_back(TierPick{i, tag}); - } else { - other_picks.push_back(TierPick{i, tag}); - } - } else { - other_picks.push_back(TierPick{i, tag}); - } - } - // Fresh-lane drain. Every queue entry is popped (its ONE first look is spent either way): kept - // if eligible AND (container-shaped OR not-yet-classified) AND room remains in the budget; - // dropped otherwise. - int fresh_room = max_count - (int)leak_picks.size(); - size_t drain_pos = fresh_queue_len <= idx_size ? idx_size - fresh_queue_len : 0; - while (!_static_anchor_fresh_queue.empty()) { - if (fresh_room <= 0) { - // Budget exhausted before the queue drained: everything remaining spends its first look now - // and falls back to the fair tiers at its index position (covered, not urgent). - _static_anchor_fresh_queue.clear(); - break; - } - jlong tag = _static_anchor_fresh_queue.front(); - _static_anchor_fresh_queue.pop_front(); - size_t pos = drain_pos; - drain_pos++; - if (pos >= idx_size || _static_anchor_index[pos] != tag) { - // The suffix-window invariant broke (cannot happen today; defensive): fall back to a search - // rather than mis-shape the entry - the queue is small, this is not a hot path once - // healthy. - auto it = - std::find(_static_anchor_index.begin(), - _static_anchor_index.end(), tag); - if (it == _static_anchor_index.end()) { - continue; - } - pos = (size_t)(it - _static_anchor_index.begin()); - } - FrontierEntry entry{}; - if (!frontier->lookupLocked(tag, &entry) || - entry.parent_tag != 0 || - (entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && - entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) || - (entry.state != FrontierEntryState::FRONTIER && - entry.state != FrontierEntryState::EXPANDED) || - isQueuedForRotation(tag) || entry.leak_tag != 0) { - continue; // dead/demoted/queued/leak-tier: first look spent, not - // fresh-kept (the leak tier selects it via the scan if it is leak-tagged) - } - bool keep = false; // container or not-yet-classified rides the - // lane; the wrapper admits one pass before reconcile can classify its - // class - if (pos < _static_anchor_own_class_tags.size()) { - auto shape_it = - _class_shape_cache.find(_static_anchor_own_class_tags[pos]); - keep = shape_it == _class_shape_cache.end() || - shape_it->second == (u8)AnchorClassShape::CONTAINER; - } else { - keep = true; // no own-class tag recorded - treat as unknown - } - if (!keep) { - continue; // classified non-container: the other tier owns it - } - fresh_picks.push_back(TierPick{pos, tag}); - fresh_kept_tags.insert(tag); - fresh_room--; - } - }); - // Cursor-fair consumption of one tier: scan picks (sorted by pos by construction) starting at - // entries with pos >= cursor, stop at `want` OR at the lap end (NO within-call wrap: re-walking - // anchors this same call already covered would waste walk budget - the leftover budget flows to - // the next tier instead, and the cursor resets to 0 so the NEXT call starts a fresh lap). - auto consume_tier_fair = [&](const std::vector &picks, - size_t &cursor, int want) { - int took = 0; - if (want <= 0 || picks.empty()) { - return took; - } - size_t consumed_pos = 0; - for (size_t k = 0; k < picks.size() && took < want; k++) { - const TierPick &p = picks[k]; - if (p.pos < cursor) { - continue; - } - if (fresh_kept_tags.count(p.tag) > 0) { - continue; - } - selected.push_back(p.tag); - consumed_pos = p.pos; - took++; - } - if (took > 0) { - cursor = consumed_pos + 1 >= idx_size ? 0 : consumed_pos + 1; - } else { - // Took nothing AND no pick sits at or ahead of the cursor: this lap - // already passed every current member of the tier (members selected in - // earlier calls and since demoted out of eligibility). Without a reset - // the cursor never wraps again - every later call skips them all - // (p.pos < cursor) and the tier starves until a NEW anchor is appended - // at a higher pos. Treat the lap as completed-but-unproductive and - // restart it, matching the wrap semantics applied above. - bool any_ahead = false; - for (const TierPick &p : picks) { - if (p.pos >= cursor) { - any_ahead = true; - break; - } - } - if (!any_ahead) { - cursor = 0; - } - } - return took; - }; - int budget_left = max_count; - for (const TierPick &p : leak_picks) { - if (budget_left <= 0) { - break; - } - selected.push_back(p.tag); - budget_left--; - } - // Fresh lane: queue order (admission order) so a burst larger than the budget spends the oldest - // first looks first and nothing jumps the queue; outranked fresh anchors fall back to the fair - // tiers at their positions (the drain already dropped them from the queue). - for (const TierPick &p : fresh_picks) { - if (budget_left <= 0) { - break; - } - selected.push_back(p.tag); - budget_left--; - } - budget_left -= consume_tier_fair(container_picks, _anchor_container_cursor, - budget_left); - // The other tier is the last consumer of the budget - its leftover has no further reader, so - // don't accumulate it back into budget_left (a dead store clang scan-build flags). - consume_tier_fair(other_picks, _anchor_other_cursor, budget_left); - return selected; -} - -void ReferenceChainTracker::pushAtRiskStaticAnchor(jlong tag, u32 klass_id) { - if (_static_anchor_fifo_set.contains(tag)) { - return; - } - if (_static_anchor_fifo.size() >= STATIC_ANCHOR_FIFO_CAP) { - // The per-class quota prevents a single class from saturating this queue. - return; - } - auto count_it = _static_anchor_fifo_klass_counts.find(klass_id); - if (count_it != _static_anchor_fifo_klass_counts.end() && - count_it->second >= STATIC_ANCHOR_ATRISK_PER_KLASS_CAP) { - // Per-class quota drop: this class already holds its share of the lane, and its oldest entry - // drains within a few passes (STATIC_ANCHOR_FIFO_DRAIN=16/pass). - return; - } - if (count_it == _static_anchor_fifo_klass_counts.end()) { - count_it = _static_anchor_fifo_klass_counts.emplace(klass_id, 0U).first; - } - count_it->second++; - _static_anchor_fifo.push_back(AtRiskAnchor{tag, klass_id}); - _static_anchor_fifo_set.insert(tag); -} - -void ReferenceChainTracker::addToStaticAnchorIndex(jlong tag, - jlong own_class_tag, - u8 root_kind) { - if (root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && - root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) { - return; - } - // Dedup: a push at first admission and another at upgrade would double-add. - if (!_static_anchor_index_tags.insert(tag).second) { - return; - } - _static_anchor_index.push_back(tag); - _static_anchor_own_class_tags.push_back(own_class_tag); - // Walk newly admitted anchors before the fair cursors. - _static_anchor_fresh_queue.push_back(tag); - if (_static_anchor_fresh_queue.size() > STATIC_ANCHOR_FRESH_CAP) { - _static_anchor_fresh_queue.pop_front(); - } -} - -bool ReferenceChainTracker::resolveContainerInterfaceTags( - jvmtiEnv *jvmti, JNIEnv *jni) { - if (_collection_iface_class_tag != 0 && _map_iface_class_tag != 0) { - return true; - } - // resolveLoadedClasses() tags every loaded class (including these bootstrap interfaces) with its - // class-tag-allocator tag (a NEGATIVE value - see nextClassTag()'s own comment) before any anchor - // can be admitted, but classify defensively: if an interface object somehow carries no tag yet, - // mint one via the shared allocator (same sequence resolveLoadedClasses() itself uses) so the - // comparison below is well-defined. - struct Iface { - const char *name; - jlong *tag_out; - }; - Iface ifaces[2] = {{"java/util/Collection", &_collection_iface_class_tag}, - {"java/util/Map", &_map_iface_class_tag}}; - for (const Iface &iface : ifaces) { - if (*iface.tag_out != 0) { - continue; - } - // Each interface is resolved independently: a transient JVMTI error on - // one of them (GetTag/SetTag/FindClass failing for exactly one) must not - // abort the other's resolution - the already-resolved tag stays cached in - // its slot either way, and a per-interface failure only leaves THAT tag - // at 0 for this call (the caller treats a false return as "shapes - // unknown this pass" and retries on the next reconcile). Failing the - // whole call on the first error would keep both anchors unclassified for - // as long as one interface keeps erroring, even though the other resolved - // fine. - jclass local = jni->FindClass(iface.name); - if (jniExceptionCheck(jni) || local == nullptr) { - jni->ExceptionClear(); - continue; - } - jlong tag = 0; - bool ok = jvmti->GetTag(local, &tag) == JVMTI_ERROR_NONE; - if (ok && tag == 0) { - jlong new_tag = nextClassTag(); - if (jvmti->SetTag(local, new_tag) == JVMTI_ERROR_NONE) { - // Adopt-on-reread, same cross-tracker race as - // resolveLoadedClasses()/mintStableClassTagIfNeeded(): another tracker may have installed - // its own tag between our GetTag and SetTag - keep the one tag the interface object - // carries. - jlong installed = 0; - if (jvmti->GetTag(local, &installed) == JVMTI_ERROR_NONE && - installed != 0) { - tag = installed; - } else { - tag = new_tag; - } - } else { - ok = false; - } - } - if (ok && tag != 0) { - *iface.tag_out = tag; - } - jni->DeleteLocalRef(local); - // A per-interface GetTag/SetTag failure leaves that tag at 0 - the final - // check below reports false only if an interface genuinely has no tag, - // never as an early abort that skips the other interface. - } - return _collection_iface_class_tag != 0 && _map_iface_class_tag != 0; -} - -bool ReferenceChainTracker::classImplementsContainerOrMap(jvmtiEnv *jvmti, - JNIEnv *jni, - jclass klass) { - // BFS over the superclass chain + every visited class's interfaces, comparing GetTag() against - // the two cached interface class tags. - std::vector work; - std::unordered_set visited; - work.push_back(klass); - bool found = false; - int hops = 0; - while (!found && !work.empty() && hops++ < 64) { - jclass cur = work.back(); - work.pop_back(); - jlong cur_tag = 0; - if (jvmti->GetTag(cur, &cur_tag) != JVMTI_ERROR_NONE || cur_tag == 0) { - // Early exit: the popped ref is this walk's own (GetSuperclass/ GetImplementedInterfaces - // local, or the caller's klass) - delete it like the loop bottom does, or the long-lived - // engine thread leaks a JNI local ref per untagged hop. - if (cur != klass) { - jni->DeleteLocalRef(cur); - } - continue; - } - if (visited.count(cur_tag) > 0) { - // Early exit: same local-ref ownership as above - delete before continuing (a hierarchy - // diamond revisits interfaces here). - if (cur != klass) { - jni->DeleteLocalRef(cur); - } - continue; - } - visited.insert(cur_tag); - if (cur_tag == _collection_iface_class_tag || - cur_tag == _map_iface_class_tag) { - found = true; - // The popped `cur` ref never reaches the loop's bottom delete. - if (cur != klass) { - jni->DeleteLocalRef(cur); - } - break; - } - jclass super = jni->GetSuperclass(cur); - if (!jniExceptionCheck(jni) && super != nullptr) { - work.push_back(super); - } else { - jni->ExceptionClear(); - } - jint iface_count = 0; - jclass *ifaces = nullptr; - if (jvmti->GetImplementedInterfaces(cur, &iface_count, &ifaces) == - JVMTI_ERROR_NONE && - ifaces != nullptr) { - for (jint i = 0; i < iface_count; i++) { - if (ifaces[i] != nullptr) { - work.push_back(ifaces[i]); - } - } - jvmti->Deallocate((unsigned char *)ifaces); - } - // `cur` is either the caller-provided klass (caller-managed ref - NOT deleted here) or a ref - // this walk minted (GetSuperclass/ GetImplementedInterfaces locals, deleted immediately after - // use). - if (cur != klass) { - jni->DeleteLocalRef(cur); - } - } - // Single exit: every remaining ref minted into `work` (early hop-bound exit or the found-break) - // is deleted here rather than leaking locals for the process lifetime (the engine thread never - // detaches). - for (jclass r : work) { - if (r != nullptr && r != klass) { - jni->DeleteLocalRef(r); - } - } - return found; -} - -void ReferenceChainTracker::reconcileAnchorClassShapes(jvmtiEnv *jvmti, - JNIEnv *jni) { - if (jni == nullptr) { - return; - } - if (_static_anchor_own_class_tags.empty()) { - return; - } - // Collect up to ANCHOR_SHAPE_RECONCILE_BUDGET distinct class tags that appear in the anchor index - // but are not yet classified. - std::vector unknown; - unknown.reserve(8); - std::unordered_set seen; - for (jlong class_tag : _static_anchor_own_class_tags) { - if (class_tag == 0 || seen.count(class_tag) > 0 || - _class_shape_cache.count(class_tag) > 0) { - continue; - } - seen.insert(class_tag); - unknown.push_back(class_tag); - if ((int)unknown.size() >= ANCHOR_SHAPE_RECONCILE_BUDGET) { - break; - } - } - if (unknown.empty()) { - return; - } - if (!resolveContainerInterfaceTags(jvmti, jni)) { - return; - } - // The per-class interface walk below mints local refs (GetSuperclass, GetImplementedInterfaces) - // that are only deleted as the BFS pops them; bound the outstanding count explicitly rather than - // relying on the JVM to grow the local-ref table. - if (jni->EnsureLocalCapacity(512) < 0 || jniExceptionCheck(jni)) { - jni->ExceptionClear(); - return; - } - // One GetObjectsWithTags call resolves the class objects for the whole batch (class objects are - // tagged with their class tags). - jint obj_count = 0; - jobject *objs = nullptr; - jlong *obj_tags = nullptr; - if (jvmti->GetObjectsWithTags((jint)unknown.size(), unknown.data(), - &obj_count, &objs, &obj_tags) != - JVMTI_ERROR_NONE || - obj_count <= 0) { - if (objs != nullptr) { - jvmti->Deallocate((unsigned char *)objs); - } - if (obj_tags != nullptr) { - jvmti->Deallocate((unsigned char *)obj_tags); - } - return; - } - for (jint i = 0; i < obj_count; i++) { - jclass klass = (jclass)objs[i]; - jlong class_tag = obj_tags[i]; - // class tags are NEGATIVE (a namespace disjoint from positive frontier tags); 0 means the - // object was never tagged - skip only that. - if (class_tag == 0 || klass == nullptr) { - if (klass != nullptr) { - jni->DeleteLocalRef(klass); - } - continue; - } - AnchorClassShape shape = classImplementsContainerOrMap(jvmti, jni, klass) - ? AnchorClassShape::CONTAINER - : AnchorClassShape::NON_CONTAINER; - _class_shape_cache[class_tag] = (u8)shape; - // GetObjectsWithTags() returned a local ref for every resolved class - this runs on the - // long-lived BFS thread, where undeleted locals accumulate until detach and pin their classes - // against unload. - jni->DeleteLocalRef(klass); - } - jvmti->Deallocate((unsigned char *)objs); - jvmti->Deallocate((unsigned char *)obj_tags); -} - -int ReferenceChainTracker::drainStaticAnchorFifo(int max_count, - std::vector &out) { - if (max_count <= 0 || _static_anchor_fifo.empty()) { - return 0; - } - int drained = 0; - while (drained < max_count && !_static_anchor_fifo.empty()) { - AtRiskAnchor entry = _static_anchor_fifo.front(); - _static_anchor_fifo.pop_front(); - auto count_it = _static_anchor_fifo_klass_counts.find(entry.klass_id); - if (count_it != _static_anchor_fifo_klass_counts.end() && - --count_it->second == 0) { - // Erased at zero so the map is bounded by the FIFO's live contents (<= 1024 distinct - // classes), not by the search lifetime. - _static_anchor_fifo_klass_counts.erase(count_it); - } - out.push_back(entry); - drained++; - } - _static_anchor_fifo_set.rebuildFrom(_static_anchor_fifo); - return drained; -} - -void ReferenceChainTracker::requeueStaticAnchorFifoFront( - const std::vector &entries) { - if (entries.empty()) { - return; - } - // Reverse order onto the front preserves the tags' relative FIFO order (push_front of the LAST - // entry first leaves the FIRST entry at the deque's front). - for (size_t i = entries.size(); i-- > 0;) { - _static_anchor_fifo_klass_counts[entries[i].klass_id]++; - _static_anchor_fifo.push_front(entries[i]); - } - _static_anchor_fifo_set.rebuildFrom(_static_anchor_fifo); -} - -void ReferenceChainTracker::walkStaticFieldAnchors( - jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &anchor_tags, - int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, - u64 *safepoint_ticks, std::vector *unwalked) { - if (anchor_tags.empty()) { - return; - } - // Resolve all anchors in one O(tag_map) call. - jint resolved_count = 0; - jobject *objects = nullptr; - jlong *resolved_tags = nullptr; - if (jvmti->GetObjectsWithTags((jint)anchor_tags.size(), anchor_tags.data(), - &resolved_count, &objects, - &resolved_tags) != JVMTI_ERROR_NONE) { - return; - } - int walked = 0; - // First index the walk did NOT consume (breaks before an anchor's walk report i, breaks after - // report i+1; a completed loop keeps the resolved_count sentinel). - jint first_unwalked = resolved_count; - // First index whose local ref has not been deleted yet. Every break path deletes objects[i] - // before breaking, so anything at or after i+1 still holds a live local ref and must be cleaned - // up below - this runs on the long-lived BFS thread, where undeleted locals accumulate until - // detach and pin their objects against collection. - jint first_undeleted = resolved_count; - for (jint i = 0; i < resolved_count; i++) { - FrontierEntry entry{}; - if (!_frontier->lookup(resolved_tags[i], &entry)) { - // Dead-or-stale between selection and here - skip; release machinery owns dead-entry cleanup, - // never here. - jni->DeleteLocalRef(objects[i]); - continue; - } - int remaining = budget - *edges_admitted; - if (remaining <= 0) { - jni->DeleteLocalRef(objects[i]); - first_unwalked = i; - first_undeleted = i + 1; - break; - } - int edges_before = *edges_admitted; - descendFromAnchor(jvmti, jni, objects[i], resolved_tags[i], entry.depth, - /*anchor_descend_class_tag=*/0, remaining, edges_admitted, - truncated, frontier_cap_hit, safepoint_ticks); - TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors anchor walk " - "outcome tag=%lld edges=%d truncated=%d cap_hit=%d", - (long long)resolved_tags[i], *edges_admitted - edges_before, - (int)*truncated, (int)*frontier_cap_hit); - walked++; - jni->DeleteLocalRef(objects[i]); - if (*truncated && !*frontier_cap_hit) { - // Budget/deadline exhausted mid-set - remaining anchors keep their rotation turn via the - // cursor next pass (the wrapping cursor already tolerates a short selection). - first_unwalked = i + 1; - first_undeleted = i + 1; - break; - } - if (*frontier_cap_hit) { - first_unwalked = i + 1; - first_undeleted = i + 1; - break; - } - } - // Release the local refs of anchors the early exits above skipped - each break only deleted its - // own objects[i]. - for (jint i = first_undeleted; i < resolved_count; i++) { - jni->DeleteLocalRef(objects[i]); - } - if (unwalked != nullptr && first_unwalked < resolved_count) { - unwalked->insert(unwalked->end(), resolved_tags + first_unwalked, - resolved_tags + resolved_count); - } - jvmti->Deallocate((unsigned char *)objects); - jvmti->Deallocate((unsigned char *)resolved_tags); - TEST_LOG_SUMMARY("ReferenceChainTracker::walkStaticFieldAnchors selected=%zu " - "walked=%d edges_admitted=%d truncated=%d frontier_cap_hit=%d", - anchor_tags.size(), walked, *edges_admitted, (int)*truncated, - (int)*frontier_cap_hit); -} - -void ReferenceChainTracker::walkCandidateThreadLocals( - jvmtiEnv *jvmti, JNIEnv *jni, int budget, int *edges_admitted, - bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks) { - if (_candidate_count <= 0) { - return; - } - // Flatten the per-slot qualifying-tid snapshot into (slot, tid) pairs, then walk up to - // THREAD_WALK_MAX_ANCHORS of them per pass, rotating via _thread_walk_anchor_cursor so every - // qualifying tid gets a turn within ceil(total / THREAD_WALK_MAX_ANCHORS) passes instead of - // always walking the first candidates' tids. - int slot[MAX_CANDIDATE_QUALIFYING_TIDS * MAX_LEAK_CANDIDATES_FROM_LT]; - jint tid[sizeof(slot) / sizeof(slot[0])]; - int total = 0; - for (int s = 0; s < _candidate_count; s++) { - for (int q = 0; q < _candidate_qualifying_tid_count[s]; q++) { - if (total >= (int)(sizeof(slot) / sizeof(slot[0]))) { - break; - } - slot[total] = s; - tid[total] = _candidate_qualifying_tids[s][q]; - total++; - } - } - if (total == 0) { - return; - } - jlong descend_class_tag = resolveThreadLocalMapClassTag(jvmti, jni); - if (_thread_walk_anchor_cursor < 0 || - _thread_walk_anchor_cursor >= total) { - _thread_walk_anchor_cursor = 0; - } - int walked = 0; - const int start = _thread_walk_anchor_cursor; - int i = start; - do { - jobject thread_obj; - { - MutexLocker ml(_thread_objects_lock); - auto it = _thread_objects.find(tid[i]); - if (it == _thread_objects.end()) { - thread_obj = nullptr; // Thread died/never registered - skip - } else { - thread_obj = it->second; - } - } - if (thread_obj != nullptr) { - // Anchor admission, idempotent across passes: a tag that still maps to a live entry is reused - // as-is (the Thread object is commonly root-attached by root enumeration already); a stale - // positive tag (search restart reissued tags from 1, releaseSearchTags() did not clear this - // object because release only touches FrontierTable entries) must be re-minted, otherwise the - // walk would parent new children onto a dead table slot or, worse, onto the entry a reissued - // tag now belongs to. - jlong anchor_tag = getTag(jvmti, thread_obj); - u32 anchor_depth = 0; - FrontierEntry anchor_entry{}; - if (anchor_tag > 0 && _frontier->lookup(anchor_tag, &anchor_entry)) { - anchor_depth = anchor_entry.depth; - } else { - jclass thread_class = jni->GetObjectClass(thread_obj); - jlong class_tag = 0; - jvmti->GetTag(thread_class, &class_tag); - u32 referrer_klass = classTags()->resolve(class_tag); - jlong fresh_tag = tagObject(jvmti, thread_obj); - if (fresh_tag != 0 && - _frontier->insert(fresh_tag, 0, referrer_klass, 0, - FrontierEntryState::FRONTIER, - (u8)JVMTI_HEAP_REFERENCE_THREAD, class_tag)) { - anchor_tag = fresh_tag; - } else { - if (fresh_tag != 0) { - // Frontier insert failed (table full): the tag-release scan only touches inserted - // frontier entries, so an installed-but-unowned tag would survive the search and - // collide with a reused tag number after a restart (fresh _next_tag from 1). - clearTag(jvmti, thread_obj); - } - anchor_tag = 0; - } - jni->DeleteLocalRef(thread_class); - } - if (anchor_tag != 0) { - int remaining = budget - *edges_admitted; - if (remaining > 0) { - descendFromAnchor(jvmti, jni, thread_obj, anchor_tag, anchor_depth, - descend_class_tag, remaining, edges_admitted, - truncated, frontier_cap_hit, safepoint_ticks); - walked++; - } - } - } - i = (i + 1) % total; - if (*frontier_cap_hit || walked >= THREAD_WALK_MAX_ANCHORS || - budget - *edges_admitted <= 0) { - break; - } - } while (i != start); - _thread_walk_anchor_cursor = i; - TEST_LOG_SUMMARY("ReferenceChainTracker::walkCandidateThreadLocals candidates=%d " - "tids=%d walked=%d edges_admitted=%d truncated=%d " - "frontier_cap_hit=%d", - _candidate_count, total, walked, *edges_admitted, (int)*truncated, - (int)*frontier_cap_hit); -} - -void ReferenceChainTracker::registerExistingThreads(jvmtiEnv *jvmti, - JNIEnv *jni) { - if (!_enabled || jvmti == nullptr || jni == nullptr) { - return; - } - // onThreadStart() cannot register threads that predate profiler attachment. - jint thread_count = 0; - jthread *thread_objects = nullptr; - if (jvmti->GetAllThreads(&thread_count, &thread_objects) != JVMTI_ERROR_NONE) { - return; - } - for (jint i = 0; i < thread_count; i++) { - jthread thread = thread_objects[i]; - if (thread == nullptr) { - continue; - } - int tid = JVMThread::nativeThreadId(jni, thread); - if (jni->ExceptionCheck()) { - jni->ExceptionClear(); - continue; - } - if (tid >= 0) { - registerThreadObject(jni, tid, thread); - } - jni->DeleteLocalRef(thread); - } - jvmti->Deallocate((unsigned char *)thread_objects); -} - -void ReferenceChainTracker::registerThreadObject(JNIEnv *jni, int tid, - jthread thread) { - if (!_enabled || jni == nullptr || thread == nullptr) { - return; - } - jobject ref = jni->NewGlobalRef(thread); - if (ref == nullptr) { - return; - } - MutexLocker ml(_thread_objects_lock); - auto it = _thread_objects.find(tid); - if (it != _thread_objects.end()) { - // Same deferred-deletion rule as unregisterThreadObject(): a walk may still hold a copy of the - // replaced ref. - _thread_refs_pending_delete.push_back(it->second); - } - _thread_objects[tid] = ref; -} - -void ReferenceChainTracker::unregisterThreadObject(JNIEnv *jni, int tid) { - if (jni == nullptr) { - return; - } - MutexLocker ml(_thread_objects_lock); - auto it = _thread_objects.find(tid); - if (it != _thread_objects.end()) { - // NOT DeleteGlobalRef() here: walkCandidateThreadLocals() may have already copied this jobject - // out of the map (lock released) and still be using it as a FollowReferences anchor - deleting - // a global ref invalidates it for every other JNI call (JNI spec), so deletion is deferred to - // releaseEndedThreadRefs() on the BFS thread (see _thread_refs_pending_delete's comment). - _thread_refs_pending_delete.push_back(it->second); - _thread_objects.erase(it); - } -} - -void ReferenceChainTracker::releaseEndedThreadRefs(JNIEnv *jni) { - if (jni == nullptr) { - return; - } - std::vector pending; - { - MutexLocker ml(_thread_objects_lock); - pending.swap(_thread_refs_pending_delete); - } - for (size_t i = 0; i < pending.size(); i++) { - jni->DeleteGlobalRef(pending[i]); - } -} - -void ReferenceChainTracker::releaseAllThreadObjects(JNIEnv *jni) { - if (jni == nullptr) { - return; - } - // Recording stop: the BFS thread is joined (Profiler::stop() order), so no walk phase can hold a - // copied ref - the deferred-deletion indirection of unregisterThreadObject() is unnecessary here - // and every ref can go now. - std::vector pending; - { - MutexLocker ml(_thread_objects_lock); - for (auto &kv : _thread_objects) { - pending.push_back(kv.second); - } - _thread_objects.clear(); - pending.insert(pending.end(), _thread_refs_pending_delete.begin(), - _thread_refs_pending_delete.end()); - _thread_refs_pending_delete.clear(); - } - for (size_t i = 0; i < pending.size(); i++) { - jni->DeleteGlobalRef(pending[i]); - } -} - diff --git a/ddprof-lib/src/main/cpp/referenceChainEvents.cpp b/ddprof-lib/src/main/cpp/referenceChainEvents.cpp deleted file mode 100644 index 648cc360bd..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainEvents.cpp +++ /dev/null @@ -1,820 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -#include "referenceChains.h" -#include "referenceChainInternal.h" -#include "common.h" -#include "counters.h" -#include "jniHelper.h" -#include "jvmThread.h" -#include "livenessTracker.h" -#include "log.h" -#include "objectSampler.h" -#include "os.h" -#include "profiler.h" -#include "rcDebugLevel.h" -#include "tsc.h" -#include "vmEntry.h" -#include -#include -#include -#include -#include -#include -#include -#include - -// Target-selection bridging step - LivenessTracker's leak-candidate ranking feeds - -void ReferenceChainTracker::requeueChainRootForRotation(jlong tag) { - if (_frontier == nullptr || tag <= 0) { - return; - } - // Walk the parent chain up to the root-attached entry - the same links reconstructChain() walks, - // but we only need the tag, not the class ids. - jlong root_tag = tag; - FrontierEntry entry{}; - int hops = 0; - while (hops++ < _hop_cap) { - if (!_frontier->lookup(root_tag, &entry) || entry.parent_tag == 0) { - break; - } - root_tag = entry.parent_tag; - } - if (root_tag == tag) { - return; // tag IS the root - nothing above it to requeue - } - if (!_frontier->lookup(root_tag, &entry) || - entry.state != FrontierEntryState::EXPANDED) { - return; // root pruned or still pending expansion - nothing to re-walk - } - if (isQueuedForRotation(root_tag) || - _priority_expand.size() >= PRIORITY_EXPAND_CAP) { - return; - } - TEST_LOG("ReferenceChainTracker::requeueChainRootForRotation root_tag=%lld " - "target_tag=%lld", - (long long)root_tag, (long long)tag); - _priority_expand.push_back(root_tag); - _priority_expand_set.insert(root_tag); -} - -namespace { - -// The discovered-chain gate's suppression predicate, shared by EVERY site that caches a resolved -// chain - the poll's discovered-instances loop AND both representative build paths (the -// canary/marker path and the normal-tag path). -bool suppressChainEvent(const ReferenceChainEvent &event) { - return event._depth < 2 && isTransientRootKind(event._root_kind); -} - -} // namespace - -// Chain-event reconstruction for a discovered/correlated instance (out of line from -// referenceChains.h so these sites share the TU's level-gated TEST_LOG; per-instance outcomes are -// level-2 diagnostics). -bool ReferenceChainTracker::buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, - jlong target_tag, - ReferenceChainEvent *out) { - if (_frontier == nullptr || out == nullptr) { - TEST_LOG("ReferenceChainTracker::buildChainEvent false: " - "frontier=%p out=%p", (void *)_frontier, (void *)out); - return false; - } - FrontierEntry entry{}; - if (!_frontier->lookup(target_tag, &entry)) { - TEST_LOG("ReferenceChainTracker::buildChainEvent false: " - "target_tag=%lld not in frontier", (long long)target_tag); - return false; - } - std::vector chain; - std::vector edges; - u8 root_kind = 0; - FrontierEntry terminal{}; - if (!_frontier->reconstructChain(target_tag, &chain, &root_kind, &edges, - &terminal)) { - TEST_LOG("ReferenceChainTracker::buildChainEvent false: " - "reconstructChain failed for target_tag=%lld", - (long long)target_tag); - return false; - } - appendStaticFieldRootType(terminal, &chain, &edges); - TEST_LOG("ReferenceChainTracker::buildChainEvent target_tag=%lld chain_size=%zu " - "chain[0]=%u depth=%u root_kind=%u leak_tag=%lld", - (long long)target_tag, chain.size(), chain.empty() ? 0u : chain[0], - entry.depth, (unsigned)root_kind, (long long)entry.leak_tag); - out->_target_tag = entry.leak_tag != 0 ? (u64)entry.leak_tag : (u64)target_tag; - out->_depth = entry.depth; - out->_root_kind = root_kind; - out->_hops.resize(chain.size()); - for (size_t i = 0; i < chain.size(); i++) { - out->_hops[i].klass_id = chain[i]; - } - // Retention-edge labels, aligned with the hops (see fillHopEdgeLabels()). - fillHopEdgeLabels(jvmti, jni, edges, &out->_hops); - return true; -} - -// Appends the root TYPE as a chain element for a static-field-rooted chain: the frontier path's -// root-side end is the static field's HOLDER instance (the object stored in the field), but the -// chain's root is the DECLARING CLASS - the holder is "the field instance referenced by the root -// type", one hop below it. -void ReferenceChainTracker::appendStaticFieldRootType( - const FrontierEntry &terminal, std::vector *chain, - std::vector *edges) { - if (chain == nullptr || - terminal.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || - terminal.referrer_class_tag == 0) { - return; - } - u32 root_klass = classTags()->resolve(terminal.referrer_class_tag); - if (root_klass == 0) { - return; - } - chain->push_back(root_klass); - if (edges != nullptr) { - ChainHopEdge root_edge{}; - root_edge.field_index = -1; - root_edge.edge_kind = terminal.root_kind; - root_edge.referrer_class_tag = 0; - edges->push_back(root_edge); - } -} - -// Canary chain reconstruction (out of line for the same reason). -bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, - ReferenceChainEvent *out) { - if (_frontier == nullptr || out == nullptr || candidate_idx < 0 || - candidate_idx >= _candidate_count) { - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " - "frontier=%p out=%p idx=%d count=%d", - (void *)_frontier, (void *)out, candidate_idx, - _candidate_count); - return false; - } - jlong parent_tag = _candidate_parent_tags[candidate_idx]; - u32 candidate_klass = _candidate_referrer_klasses[candidate_idx]; - jlong frontier_tag = _candidate_frontier_tags[candidate_idx]; - std::vector chain; - u8 root_kind = 0; - // The root-attached entry the walk ends at - both branches below leave `entry` holding it (the - // walk's last lookup, or the candidate's own entry for a root-referenced candidate). - FrontierEntry terminal{}; - if (parent_tag > 0) { - // Walk parent_tag back to root through the frontier table. - FrontierEntry entry{}; - if (!_frontier->lookup(parent_tag, &entry)) { - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " - "parent_tag=%lld not in frontier (candidate=%d)", - (long long)parent_tag, candidate_idx); - return false; - } - // Bounded like every sibling walk over this same parent chain: - // FrontierTable::reconstructChain() bounds at maxCapacity() hops and - // returns false on a "cyclic or corrupt parent chain", - // requeueChainRootForRotation() bounds at _hop_cap, and improveChain()'s - // cycle guard bounds at 4096. improveChain() structurally prevents cycles - // (it refuses a parent whose chain routes through the entry), but the - // table's contents are also written by insert() with no such validation, - // so a corrupt chain must fail safe instead of spinning this poll-thread - // walk forever: every tag maps to a distinct slot (tags are never - // reused), so a well-formed chain can visit at most maxCapacity() entries - // before reaching parent_tag == 0 or repeating a slot. - const int hop_bound = _frontier->maxCapacity(); - int hops = 0; - for (jlong tag = parent_tag; tag > 0; hops++) { - if (hops > hop_bound) { - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " - "chain walk exceeded hop bound (%d) - cyclic or " - "corrupt parent chain (candidate=%d)", - hops, candidate_idx); - return false; - } - if (!_frontier->lookup(tag, &entry)) { - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " - "chain walk: tag=%lld not in frontier (candidate=%d)", - (long long)tag, candidate_idx); - return false; - } - chain.push_back(entry.referrer_klass); - tag = entry.parent_tag; - } - // The root kind describes the chain's ROOT, not the candidate-side parent: - // the walk's last iteration is always the root-attached entry (parent_tag - // == 0, root_kind != 0), so `entry` holds it here - same terminal-root - // semantics reconstructChain() uses for *out_root_kind. The parent-side - // entry read before the loop would almost always yield 0 (interior entries - // carry root_kind == 0), misreporting every walk-reconstructed chain and - // defeating suppressChainEvent()'s transient-root gate. - root_kind = entry.root_kind; - terminal = entry; - } else if (parent_tag == 0 && frontier_tag > 0) { - // Root-referenced candidate: chain is just [candidate_klass]. - FrontierEntry entry{}; - if (!_frontier->lookup(frontier_tag, &entry)) { - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " - "frontier_tag=%lld not in frontier (candidate=%d)", - (long long)frontier_tag, candidate_idx); - return false; - } - root_kind = entry.root_kind; - terminal = entry; - } else { - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " - "never pruned (candidate=%d parent_tag=%lld frontier_tag=%lld)", - candidate_idx, (long long)parent_tag, - (long long)frontier_tag); - return false; // never pruned (candidate not reached) - } - // Prepend the candidate's own referrer_klass. - chain.push_back(candidate_klass); - // The chain was built root-to-parent; reverse to get candidate-to-root. - std::reverse(chain.begin(), chain.end()); - // Same root-type element buildChainEvent() appends: the canary walk's terminal entry is the - // root-attached entry, and for a static-field root the declaring class belongs at the chain's - // root-side end (after the reverse). - appendStaticFieldRootType(terminal, &chain, nullptr); - out->_target_tag = (u64)frontier_tag; - out->_depth = _candidate_depths[candidate_idx]; - out->_root_kind = root_kind; - const size_t chain_size = chain.size(); - out->_hops.resize(chain_size); - for (size_t i = 0; i < chain.size(); i++) { - out->_hops[i].klass_id = chain[i]; - } - TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent candidate=%d " - "parent_tag=%lld chain_size=%zu", - candidate_idx, (long long)parent_tag, chain_size); - return true; -} - -void ReferenceChainTracker::pollWatchedTargets(jvmtiEnv *jvmti, JNIEnv *jni) { - if (!_enabled || jvmti == nullptr || jni == nullptr || - !LivenessTracker::instance()->gcGenerationsEnabled()) { - // Avoid candidate-table work when generation tracking is disabled. - return; - } - - // Stamp every entry this poll refreshes with the current search generation. - const u64 current_search_ns = load(_search_start_ns); - - // klass_ids resolved (and therefore already pruned-if-dead) by the candidate loop below, so the - // prune pass afterwards skips re-resolving them - it only needs to cover cached klasses that are - // no longer flagged. - - // selectLeakCandidates() clamps this to its private candidate limit. - constexpr int kMaxWatchedCandidates = 8; - KlassCandidate candidates[kMaxWatchedCandidates]; - int candidate_count = LivenessTracker::instance()->selectLeakCandidates( - candidates, kMaxWatchedCandidates); - - // Publish this poll's qualifying tids as LivenessTracker's watched-admission set (see - // noteSelectedCandidates()'s own comment, livenessTracker.h): exactly the (klass, tid) scope - // tagLeakInstances() tags and this chase intercepts gets its allocations admitted at 100% instead - // of the default 10% ratio lottery. - LivenessTracker::instance()->noteSelectedCandidates(candidates, - candidate_count); - - // Refresh the faster, un-hysteresis-gated klass_id ranking rotation priority uses (see - // _watched_leak_klass_ids' own comment) - but only once selectLeakCandidates() above has ALREADY - // found at least one qualifying candidate via its own slower hysteresis gate: this mechanism is - // meant to crank once the trend detector has triggered, not to run the ranking independently - // before that gate has ever fired. - if (candidate_count > 0) { - // Snapshot the OLD watched set before overwriting it, so any klass_id that's newly appearing - // this refresh can get its one-time retroactive catch-up - // (seedLeakAccumulationForNewlyWatchedKlass() - see _watched_leak_klass_ids' own comment for - // why admission-time tracking alone cannot see objects admitted before watching started). - u32 previously_watched[MAX_WATCHED_LEAK_KLASSES]; - int previously_watched_count = _watched_leak_klass_count; - for (int i = 0; i < previously_watched_count; i++) { - previously_watched[i] = _watched_leak_klass_ids[i]; - } - _watched_leak_klass_count = LivenessTracker::instance()->topKlassesByGenerationCount( - _watched_leak_klass_ids, MAX_WATCHED_LEAK_KLASSES); - for (int i = 0; i < _watched_leak_klass_count; i++) { - u32 klass_id = _watched_leak_klass_ids[i]; - bool already_watched = false; - for (int j = 0; j < previously_watched_count; j++) { - if (previously_watched[j] == klass_id) { - already_watched = true; - break; - } - } - if (!already_watched) { - seedLeakAccumulationForNewlyWatchedKlass(klass_id); - } - } - } - // Only log when there are candidates to act on - this poll runs on every BFS-thread wake (once - // per second), so logging a zero count is per-second noise for the common idle case. - if (candidate_count > 0) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate_count=%d", candidate_count); - // Admit any candidate selectLeakCandidates() returns this poll that doesn't already occupy a - // slot, into the next free slot. - for (int i = 0; i < candidate_count; i++) { - u32 klass_id = candidates[i].klass_id; - bool already_tracked = false; - for (int s = 0; s < _candidate_count; s++) { - if (_candidate_klass_ids[s] == klass_id) { - already_tracked = true; - break; - } - } - if (already_tracked) { - continue; - } - if (_candidate_count >= MAX_LEAK_CANDIDATES_FROM_LT) { - TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary: klass_id=%u " - "qualifies but all %d slots are occupied - not tracked this search", - klass_id, MAX_LEAK_CANDIDATES_FROM_LT); - continue; - } - // Candidate admission: the marker->leak-tag migration retired - // pre-tagging the representative object (the retired marker-tag decode - // branches are gone); this candidate is discovered when the walk or the - // poll intercepts one of its leak-tagged instances. - int slot = _candidate_count; - _candidate_klass_ids[slot] = klass_id; - _candidate_count = slot + 1; - TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary: admitted klass_id=%u " - "into slot=%d (candidate_count now %d)", - klass_id, slot, _candidate_count); - Counters::increment(REFERENCE_CHAIN_CANDIDATE_COUNT, 1); - } - // Refresh the per-slot qualifying-tid snapshot the walk phases read - // (walkCandidateThreadLocals()): zero every slot first, then fill from THIS poll's candidates - - // a klass whose per-tid trend stopped qualifying must stop having its tids walked, exactly like - // it stops consuming pool tags (tagLeakInstances() below keeps the same per-poll-candidates - // scope for the same reason). - memset(_candidate_qualifying_tid_count, 0, - sizeof(_candidate_qualifying_tid_count)); - for (int i = 0; i < candidate_count; i++) { - for (int s = 0; s < _candidate_count; s++) { - if (_candidate_klass_ids[s] != candidates[i].klass_id) { - continue; - } - int n = candidates[i].qualifying_tid_count; - if (n > MAX_CANDIDATE_QUALIFYING_TIDS) { - n = MAX_CANDIDATE_QUALIFYING_TIDS; - } - for (int q = 0; q < n; q++) { - _candidate_qualifying_tids[s][q] = candidates[i].qualifying_tids[q]; - } - _candidate_qualifying_tid_count[s] = n; - break; - } - } - // Tag the tracked instances of THIS poll's candidates with leak tags. - int tagged = LivenessTracker::instance()->tagLeakInstances( - jvmti, candidates, candidate_count); - _leak_tags_assigned = tagged; - _leak_tags_resolved = 0; // reset on each tagging round - TEST_LOG("ReferenceChainTracker::pollWatchedTargets tagLeakInstances tagged=%d", - tagged); - } - - for (int i = 0; i < candidate_count; i++) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u", i, - candidates[i].klass_id); - // Deliberately does NOT resolve candidates[i].representative directly: that field is a snapshot - // taken under selectLeakCandidates()'s own shared-lock scan, which can go stale (LRU-evicted - // and DeleteWeakGlobalRef()'d by LivenessTracker's cleanup_table(), running concurrently on a - // different thread) at any point between that call and this one - see selectLeakCandidates()'s - // comment (livenessTracker.h) for why resolving it here would be undefined behavior, not just a - // null result. - const u32 klass_id = candidates[i].klass_id; - jobject obj = LivenessTracker::instance()->resolveCandidateRepresentative( - jni, klass_id); - if (obj == nullptr) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u " - "representative could not be resolved (died/evicted)", - i, klass_id); - // The representative died, but the canary chain (if the candidate was pruned by BFS before - // the representative died) only needs the frontier table — not the live representative. - bool built_from_canary = false; - for (int s = 0; s < _candidate_count; s++) { - if (_candidate_klass_ids[s] != klass_id) continue; - if ((_candidate_found_bits & (1ULL << s)) && - _candidate_frontier_tags[s] != 0) { - jlong canary_ftag = _candidate_frontier_tags[s]; - _resolved_chains_lock.lock(); - bool need = (_resolved_chains.find(canary_ftag) == _resolved_chains.end()); - _resolved_chains_lock.unlock(); - if (need) { - ReferenceChainEvent event; - built_from_canary = buildCanaryChainEvent(s, &event); - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "buildCanaryChainEvent(dead rep, slot=%d) -> %d", - s, (int)built_from_canary); - if (built_from_canary && suppressChainEvent(event)) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "filtered depth=%u root_kind=%d canary_ftag=%lld " - "klass_id=%u (dead-rep path)", - event._depth, (int)event._root_kind, - (long long)canary_ftag, klass_id); - built_from_canary = false; - invalidateResolvedChain(canary_ftag); - } else if (built_from_canary) { - event._start_time = TSC::ticks(); - cacheResolvedChain(canary_ftag, std::move(event), - canary_ftag, current_search_ns); - } - } - } - break; - } - if (!built_from_canary) { - // The representative died. Per-instance caching means we don't erase by klass_id — chains - // for other instances of this class may still be valid. - } - continue; // candidate died, or was evicted, since LivenessTracker flagged it - } - - { - jclass obj_klass = jni->GetObjectClass(obj); - char *obj_class_name = nullptr; - if (obj_klass != nullptr && - jvmti->GetClassSignature(obj_klass, &obj_class_name, nullptr) == - JVMTI_ERROR_NONE && - obj_class_name != nullptr) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] " - "klass_id=%u class_name=%s", - i, klass_id, obj_class_name); - jvmti->Deallocate((unsigned char *)obj_class_name); - } - if (obj_klass != nullptr) { - jni->DeleteLocalRef(obj_klass); - } - } - - // Read the existing tag; seeding here would bypass the forward walk. - jlong tag = getTag(jvmti, obj); - - // NOTE: the retired canary marker-tag decode used to live here (an object - // pre-tagged with MARKER_TAG_BASE - slot took the legacy canary chain - // reconstruction). The marker->leak-tag migration stopped pre-tagging - // candidate representatives entirely - no JVMTI tag in the process can - // ever be <= MARKER_TAG_BASE (-2^62): leak tags are positive - // (LEAK_TAG_BASE), frontier tags positive, class tags small negative - // magnitudes - so the branch was unreachable, and a stale negative tag - // (a class tag) could never silently take the legacy canary path. - // Candidate chain reconstruction now runs through the leak-tag discovery - // block below and the dead-representative canary path earlier in this - // loop. - - // Normal path: tag > 0 means the walk visited this object and assigned it a - // frontier tag. - - // Keep the holder chain's root warm in the rotation queue: a growing container's current - // internals are only reachable via the holder's re-walk (requeueChainRootForRotation()'s own - // comment). - if (tag > 0) { - requeueChainRootForRotation(tag); - } - - // Reconstruct only when this klass has no current chain cached: either nothing cached yet, or - // what is cached was built from a different tag or an earlier search generation (see - // current_search_ns above). - bool need_refresh = false; - jlong rep_chain_key = 0; - for (int s = 0; s < _candidate_count; s++) { - if (_candidate_klass_ids[s] == klass_id) { - rep_chain_key = _candidate_frontier_tags[s]; - break; - } - } - if (rep_chain_key != 0) { - _resolved_chains_lock.lock(); - auto it = _resolved_chains.find(rep_chain_key); - need_refresh = (it == _resolved_chains.end() || - it->second.source_search_ns != current_search_ns); - _resolved_chains_lock.unlock(); - } - TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u tag=%lld " - "needRefresh=%d", - i, klass_id, (long long)tag, need_refresh); - if (need_refresh) { - ReferenceChainEvent event; - bool built = buildChainEvent(jvmti, jni, tag, &event); - TEST_LOG("ReferenceChainTracker::pollWatchedTargets buildChainEvent(tag=%lld) -> %d", - (long long)tag, built); - if (built && suppressChainEvent(event)) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "filtered depth=%u root_kind=%d rep_tag=%lld klass_id=%u " - "(representative path)", - event._depth, (int)event._root_kind, (long long)tag, - klass_id); - built = false; - invalidateResolvedChain(tag); - } - if (built) { - // Provisional stamp; drainPendingChainEvents() re-stamps each copy at dump time so the - // event lands in that chunk's window. - event._start_time = TSC::ticks(); - cacheResolvedChain(tag, std::move(event), tag, current_search_ns); - } - } - // tag == 0: The representative object has no tag — the BFS walk hasn't reached it yet AND it is - // not yet leak-tagged. - - // Build chain events for auto-marked discovered instances of this class. - buildDiscoveredInstanceChains(jvmti, jni, klass_id, current_search_ns); - - jni->DeleteLocalRef(obj); - } - - // Orphan fix: slots whose klass is NOT among this poll's candidates. - for (int s = 0; s < _candidate_count; s++) { - u32 slot_klass = _candidate_klass_ids[s]; - bool in_poll = false; - for (int i = 0; i < candidate_count; i++) { - if (candidates[i].klass_id == slot_klass) { - in_poll = true; - break; - } - } - if (!in_poll && slot_klass != 0) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets orphan slot " - "sweep: slot=%d klass_id=%u not in poll candidates - building " - "its discovered chains", - s, slot_klass); - buildDiscoveredInstanceChains(jvmti, jni, slot_klass, current_search_ns); - } - } - - // Per-instance caching: chains are keyed by frontier tag, not klass_id. -} - -// Inserts or refreshes klass_id's resolved chain - see _resolved_chains' comment -// (referenceChains.h) for why a resolved chain is cached and re-emitted rather than emitted once. -bool ReferenceChainTracker::cacheResolvedChain(jlong source_tag, - ReferenceChainEvent &&event, - jlong source_tag_val, - u64 source_search_ns) { - _resolved_chains_lock.lock(); - auto it = _resolved_chains.find(source_tag); - if (it == _resolved_chains.end() && - (int)_resolved_chains.size() >= MAX_RESOLVED_CHAINS) { - _resolved_chains_lock.unlock(); - Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); - TEST_LOG("ReferenceChainTracker::cacheResolvedChain dropped new source_tag=%lld, " - "cache full (at MAX_RESOLVED_CHAINS=%d)", - (long long)source_tag, MAX_RESOLVED_CHAINS); - return false; - } - CachedChain &slot = _resolved_chains[source_tag]; - slot.event = std::move(event); - slot.source_tag = source_tag_val; - slot.source_search_ns = source_search_ns; - TEST_LOG("ReferenceChainTracker::cacheResolvedChain source_tag=%lld cache_size=%d", - (long long)source_tag, (int)_resolved_chains.size()); - _resolved_chains_lock.unlock(); - return true; -} - -void ReferenceChainTracker::invalidateResolvedChain(jlong source_tag) { - _resolved_chains_lock.lock(); - auto it = _resolved_chains.find(source_tag); - if (it != _resolved_chains.end()) { - _resolved_chains.erase(it); - TEST_LOG("ReferenceChainTracker::invalidateResolvedChain source_tag=%lld", - (long long)source_tag); - } - _resolved_chains_lock.unlock(); -} - -// Builds and caches chain events for every auto-marked discovered instance recorded against a slot -// holding klass_id (see the auto-mark block in heapReferenceCallback() for how instances get -// recorded). -void ReferenceChainTracker::buildDiscoveredInstanceChains(jvmtiEnv *jvmti, - JNIEnv *jni, - u32 klass_id, - u64 current_search_ns) { -for (int s = 0; s < _candidate_count; s++) { - if (_candidate_klass_ids[s] != klass_id) continue; - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "discovered loop: klass_id=%u slot=%d discovered_count=%d", - klass_id, s, _candidate_discovered_count[s]); - if (_candidate_discovered_count[s] == 0) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "no discovered instances for klass_id=%u slot=%d", - klass_id, s); - } - for (int d = 0; d < _candidate_discovered_count[s]; d++) { - jlong disc_tag = _candidate_discovered_tags[s][d]; - if (disc_tag == 0) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "disc_tag=0 at idx=%d for klass_id=%u slot=%d", - d, klass_id, s); - continue; - } - // Skip if already cached for this instance - but only for the CURRENT search generation: - // restartSearch() resets the frontier tag namespace, so a cached entry under the same numeric - // tag from an earlier search describes a different object and must not suppress the rebuild - // (the generation check mirrors the rep-refresh paths in pollWatchedTargets()). - const u64 current_search_ns = load(_search_start_ns); - _resolved_chains_lock.lock(); - auto cached_it = _resolved_chains.find(disc_tag); - bool already_cached = (cached_it != _resolved_chains.end() && - cached_it->second.source_search_ns == - current_search_ns); - _resolved_chains_lock.unlock(); - if (already_cached) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "already_cached disc_tag=%lld klass_id=%u slot=%d idx=%d", - (long long)disc_tag, klass_id, s, d); - continue; - } - ReferenceChainEvent event; - bool built = buildChainEvent(jvmti, jni, disc_tag, &event); - // Retention-explanation filter. Only applies to discovered instances, not canary. - if (built && suppressChainEvent(event)) { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "filtered depth=%u root_kind=%d disc_tag=%lld klass_id=%u", - event._depth, (int)event._root_kind, - (long long)disc_tag, klass_id); - built = false; - // Also drop any chain cached for this tag before the filter existed (or before an - // improveChain/reparent upgraded it) - drainPendingChainEvents() re-emits cached chains - // unconditionally, so suppressing only the build would leave the noise chains re-emitting - // forever. - invalidateResolvedChain(disc_tag); - } - if (built) { - event._start_time = TSC::ticks(); - // Coverage accounting below must only advance for a chain that was actually stored - a - // cache-full drop would let the search report the candidate as found without ever emitting - // its chain. - if (cacheResolvedChain(disc_tag, std::move(event), disc_tag, - current_search_ns)) { - // Track coverage for adaptive CPU budget - if (event._target_tag >= (u64)LEAK_TAG_BASE) { - _leak_tags_resolved++; - // A qualifying thread must discover at least one candidate instance. - if (!(_candidate_found_bits & (1ULL << s))) { - _candidate_found_bits |= (1ULL << s); - _candidate_frontier_tags[s] = disc_tag; - _candidate_parent_tags[s] = 0; - _candidate_depths[s] = event._depth; - _candidate_referrer_klasses[s] = klass_id; - TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary " - "found: klass_id=%u slot=%d leak chain target_tag=%llu " - "via disc_tag=%lld (%d/%d candidates found)", - klass_id, s, (unsigned long long)event._target_tag, - (long long)disc_tag, - __builtin_popcountll(_candidate_found_bits), - _candidate_count); - } - } - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "auto-marked chain for klass_id=%u tag=%lld target_tag=%llu", - klass_id, (long long)disc_tag, - (unsigned long long)event._target_tag); - } - } else { - TEST_LOG("ReferenceChainTracker::pollWatchedTargets " - "buildChainEvent failed for discovered tag=%lld " - "klass_id=%u slot=%d disc_idx=%d", - (long long)disc_tag, klass_id, s, d); - } - } - break; -} -} - -void ReferenceChainTracker::recordDiscoveredInstance(u32 klass_id, - jlong frontier_tag, - bool leak_correlated) { - // See the declaration's own comment (referenceChains.h) for the noise-eviction rationale. - for (int s = 0; s < _candidate_count; s++) { - if (_candidate_klass_ids[s] != klass_id) { - continue; - } - if (_candidate_discovered_count[s] < MAX_DISCOVERED_INSTANCES_PER_CLASS) { - _candidate_discovered_tags[s][_candidate_discovered_count[s]++] = - frontier_tag; - TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance slot=%d " - "klass_id=%u tag=%lld leak_correlated=%d count=%d", - s, klass_id, (long long)frontier_tag, (int)leak_correlated, - _candidate_discovered_count[s]); - return; - } - if (!leak_correlated) { - return; // full - noise never displaces anything - } - // All slots full and this instance is leak-correlated: evict the first slot held by an entry - // with no leak tag (a noise instance). - for (int d = 0; d < _candidate_discovered_count[s]; d++) { - jlong victim = _candidate_discovered_tags[s][d]; - FrontierEntry victim_entry{}; - if (_frontier == nullptr || - !_frontier->lookup(victim, &victim_entry) || - victim_entry.leak_tag == 0) { - _candidate_discovered_tags[s][d] = frontier_tag; - invalidateResolvedChain(victim); - TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance evicted " - "noise slot=%d idx=%d victim_tag=%lld for leak tag=%lld", - s, d, (long long)victim, (long long)frontier_tag); - return; - } - } - TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance all slots " - "leak-correlated, dropping tag=%lld klass_id=%u", - (long long)frontier_tag, klass_id); - return; - } -} - -bool ReferenceChainTracker::correlateAdmittedLeakTag(jlong frontier_tag, - jlong leak_tag, - u32 klass_id) { - // See the declaration's own comment (referenceChains.h). Called from - // LivenessTracker::tagLeakInstances() on this same thread (pollWatchedTargets -> - // tagLeakInstances), so _candidate_* slot access here never races heapReferenceCallback's - // auto-mark path. - if (_frontier == nullptr) { - return false; - } - FrontierEntry entry{}; - if (!_frontier->lookup(frontier_tag, &entry)) { - return false; // not a live frontier tag (or the search restarted) - } - if (entry.leak_tag != 0) { - // Already correlated (idempotent) - e.g. a second tagLeakInstances round after a post-restart - // re-admission. - return true; - } - _frontier->setLeakTag(frontier_tag, leak_tag); - TEST_LOG_SUMMARY("ReferenceChainTracker::correlateAdmittedLeakTag " - "frontier_tag=%lld leak_tag=%lld depth=%u parent_tag=%lld", - (long long)frontier_tag, (long long)leak_tag, entry.depth, - (long long)entry.parent_tag); - recordDiscoveredInstance(klass_id, frontier_tag, true); - return true; -} - -void ReferenceChainTracker::drainPendingChainEvents( - std::vector *out) { - if (out == nullptr) { - return; - } - // Snapshot-and-keep, not a drain: every cached chain is copied out (and re-stamped so it lands in - // the dumping chunk's window) while the cache itself is left intact, so the same live sample's - // chain re-emits into every chunk it survives into (see _resolved_chains' comment). - u64 now = TSC::ticks(); - _resolved_chains_lock.lock(); - for (const auto &kv : _resolved_chains) { - out->push_back(kv.second.event); - out->back()._start_time = now; - } - _resolved_chains_lock.unlock(); - TEST_LOG_SUMMARY("ReferenceChainTracker::drainPendingChainEvents re-emitted=%d", - (int)out->size()); -} - -void ReferenceChainTracker::enqueuePendingAbandonedEvent() { - // Called right after runPass() (referenceChains.cpp) writes SearchState::ABANDONED, on the same - // thread, before shouldRunPass() gets a chance to call restartSearch() - so - // buildAbandonedEvent()'s live read of _search_state/_abandon_reason/etc. - ReferenceChainAbandonedEvent event; - if (!buildAbandonedEvent(&event)) { - return; - } - // Stamp when the search actually stopped, not when a later dump writes the queued event - an - // abandon is a point-in-time occurrence and dump() can lag it by a whole chunk rotation. - event._start_time = TSC::ticks(); - _pending_abandoned_events_lock.lock(); - if ((int)_pending_abandoned_events.size() >= MAX_PENDING_ABANDONED_EVENTS) { - _pending_abandoned_events_lock.unlock(); - Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); - TEST_LOG_SUMMARY("ReferenceChainTracker::enqueuePendingAbandonedEvent dropped, " - "queue full (at MAX_PENDING_ABANDONED_EVENTS=%d)", - MAX_PENDING_ABANDONED_EVENTS); - return; - } - _pending_abandoned_events.push_back(event); - TEST_LOG_SUMMARY("ReferenceChainTracker::enqueuePendingAbandonedEvent reason=%d " - "queue_size=%d", - (int)event._reason, (int)_pending_abandoned_events.size()); - _pending_abandoned_events_lock.unlock(); -} - -void ReferenceChainTracker::drainPendingAbandonedEvents( - std::vector *out) { - if (out == nullptr) { - return; - } - // True drain, unlike drainPendingChainEvents() above: each queued event describes a discrete past - // occurrence, not an ongoing live sample, so once Profiler::dump() (profiler.cpp) has emitted it - // there is nothing left to re-report on the next dump. - _pending_abandoned_events_lock.lock(); - out->insert(out->end(), _pending_abandoned_events.begin(), - _pending_abandoned_events.end()); - _pending_abandoned_events.clear(); - _pending_abandoned_events_lock.unlock(); - TEST_LOG_SUMMARY("ReferenceChainTracker::drainPendingAbandonedEvents drained=%d", - (int)out->size()); -} diff --git a/ddprof-lib/src/main/cpp/referenceChainFrontier.cpp b/ddprof-lib/src/main/cpp/referenceChainFrontier.cpp deleted file mode 100644 index a8d709c6ff..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainFrontier.cpp +++ /dev/null @@ -1,426 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -#include "referenceChainFrontier.h" -#include "counters.h" -#include "log.h" -#include "rcDebugLevel.h" -#include -#include -#include -#include -#include - -FrontierTable::FrontierTable(int max_cap) - : _table_size(0), _table_cap(0), _table_max_cap(std::max(max_cap, 0)), - _table(nullptr) { - _table_cap = std::min(INITIAL_TABLE_CAPACITY, _table_max_cap); - if (_table_cap > 0) { - _table = (FrontierEntry *)calloc(_table_cap, sizeof(FrontierEntry)); - if (_table == nullptr) { - _table_cap = 0; - } - } - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, - (jlong)_table_cap * sizeof(FrontierEntry)); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, _table_cap); -} - -FrontierTable::~FrontierTable() { - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, - -(jlong)_table_cap * sizeof(FrontierEntry)); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, -_table_cap); - free(_table); -} - -void FrontierTable::resetCapacityForTest(int max_cap) { - _table_lock.lock(); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, - -(jlong)_table_cap * sizeof(FrontierEntry)); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, -_table_cap); - free(_table); - _table = nullptr; - _table_max_cap = std::max(max_cap, 0); - _table_cap = std::min(INITIAL_TABLE_CAPACITY, _table_max_cap); - if (_table_cap > 0) { - _table = (FrontierEntry *)calloc(_table_cap, sizeof(FrontierEntry)); - if (_table == nullptr) { - _table_cap = 0; - } - } - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, - (jlong)_table_cap * sizeof(FrontierEntry)); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, _table_cap); - _table_size.store(0, std::memory_order_relaxed); - _table_lock.unlock(); -} - -bool FrontierTable::growLocked(int required_cap) { - if (required_cap <= _table_cap) { - return true; - } - if (_table_cap >= _table_max_cap) { - return false; - } - - int newcap = _table_cap; - while (newcap < required_cap && newcap < _table_max_cap) { - newcap = newcap == 0 ? std::min(INITIAL_TABLE_CAPACITY, _table_max_cap) - : std::min(newcap * 2, _table_max_cap); - } - if (newcap <= _table_cap) { - return false; - } - - FrontierEntry *tmp = - (FrontierEntry *)realloc(_table, sizeof(FrontierEntry) * newcap); - if (tmp == nullptr) { - Log::debug( - "ReferenceChains: frontier table resize to %d entries failed", newcap); - return false; - } - // realloc() does not zero the newly grown region - clear it so lookup() never returns garbage - // state for a slot that hasn't been inserted yet. - memset(tmp + _table_cap, 0, sizeof(FrontierEntry) * (newcap - _table_cap)); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, - (jlong)(newcap - _table_cap) * sizeof(FrontierEntry)); - Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, - newcap - _table_cap); - _table = tmp; - _table_cap = newcap; - return _table_cap >= required_cap; -} - -bool FrontierTable::insert(jlong tag, jlong parent_tag, u32 referrer_klass, - u32 depth, u8 state, u8 root_kind, - jlong class_tag, jint referrer_field_index, - u8 edge_kind, jlong referrer_class_tag) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return false; - } - int idx = (int)(tag - 1); - - // Exclusive lock for the whole write (growLocked() already requires it) - a shared lock here - // would not exclude lookup()'s own shared-mode read of the same slot, letting a concurrent reader - // observe a torn entry. - _table_lock.lock(); - if (idx >= _table_cap && !growLocked(idx + 1)) { - _table_lock.unlock(); - Log::debug("ReferenceChains: frontier table capacity exhausted " - "(cap=%d, max=%d, tag=%lld)", - _table_cap, _table_max_cap, (long long)tag); - return false; - } - _table[idx].parent_tag = parent_tag; - _table[idx].referrer_klass = referrer_klass; - _table[idx].depth = depth; - _table[idx].state = state; - _table[idx].root_kind = root_kind; - _table[idx].class_tag = class_tag; - _table[idx].leak_tag = 0; - _table[idx].referrer_field_index = referrer_field_index; - _table[idx].edge_kind = edge_kind; - _table[idx].referrer_class_tag = referrer_class_tag; - - // Published under the same exclusive lock as the slot write: advancing - // _table_size only after unlock lets a concurrent insert for a higher index - // CAS the size past this entry's idx first, and a shared-lock reader then - // passes its `idx < _table_size` check on a value published outside the - // lock - reading this slot before its write is guaranteed visible. Keeping - // the write+publish pair inside the lock makes the size an exact bound on - // fully-written slots for every lock-ordered reader. - int sz = _table_size.load(std::memory_order_relaxed); - while (sz < idx + 1 && - !_table_size.compare_exchange_weak(sz, idx + 1, - std::memory_order_relaxed)) { - // sz reloaded with the current value by compare_exchange_weak on failure; retry until either - // this thread wins or another thread already advanced _table_size past idx + 1. - } - _table_lock.unlock(); - return true; -} - -bool FrontierTable::lookup(jlong tag, FrontierEntry *out) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return false; - } - int idx = (int)(tag - 1); - - bool found = false; - _table_lock.lockShared(); - if (idx < _table_size) { - *out = _table[idx]; - found = true; - } - _table_lock.unlockShared(); - return found; -} - -bool FrontierTable::lookupLocked(jlong tag, FrontierEntry *out) const { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return false; - } - int idx = (int)(tag - 1); - if (idx < _table_size) { - *out = _table[idx]; - return true; - } - return false; -} - -void FrontierTable::clear(jlong tag) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return; - } - int idx = (int)(tag - 1); - - // Exclusive lock: this mutates a slot lookup() may be reading concurrently under its own shared - // lock (see insert()'s own comment above). - _table_lock.lock(); - if (idx < _table_size) { - _table[idx].state = FrontierEntryState::ABANDONED; - } - _table_lock.unlock(); -} - -void FrontierTable::markEdge(jlong tag) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return; - } - int idx = (int)(tag - 1); - - _table_lock.lock(); - if (idx < _table_size) { - _table[idx].state = FrontierEntryState::EDGE; - } - _table_lock.unlock(); -} - -void FrontierTable::markExpanded(jlong tag) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return; - } - int idx = (int)(tag - 1); - - _table_lock.lock(); - if (idx < _table_size) { - _table[idx].state = FrontierEntryState::EXPANDED; - } - _table_lock.unlock(); -} - -void FrontierTable::updateRootKind(jlong tag, u8 root_kind) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return; - } - int idx = (int)(tag - 1); - - _table_lock.lock(); - if (idx < _table_size) { - _table[idx].root_kind = root_kind; - } - _table_lock.unlock(); -} - -bool FrontierTable::improveChain(jlong tag, jlong parent_tag, - u32 referrer_klass, u32 depth, - u8 root_kind, jint referrer_field_index, - u8 edge_kind, jlong referrer_class_tag) { - // Replace a shallow root-attached entry (parent_tag == 0, depth == 0) with a deeper - // chain-attached entry when the object is reached via a longer path. - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return false; - } - // a "chain" whose parent is the entry itself is never an improvement - it is the self-edge a - // this-field produces, and it is REAL in the heap: every java.util.Collections$Synchronized* - // holder carries mutex == this, so walking such a holder's own subtree (the rotation anchor walk - // or a BFS descent) re-reports the holder as its own child through that field. - if (parent_tag == tag) { - return false; - } - // Ancestor-walk bound for the cycle guard below - see its comment. - static constexpr int IMPROVE_CHAIN_GUARD_MAX_HOPS = 4096; - { - int guard_hops = 0; - jlong cur = parent_tag; - while (cur > 0 && guard_hops <= IMPROVE_CHAIN_GUARD_MAX_HOPS) { - if (cur == tag) { - TEST_LOG_SUMMARY("FrontierTable::improveChain refused: new parent " - "chain routes through the entry (cycle) tag=%lld " - "parent_tag=%lld depth=%u", - (long long)tag, (long long)parent_tag, depth); - return false; - } - FrontierEntry guard_entry{}; - if (!lookup(cur, &guard_entry)) { - break; - } - cur = guard_entry.parent_tag; - guard_hops++; - } - if (cur != 0) { - // The parent chain neither reached a root nor was fully verified within the guard bound - - // applying this improve could embed an unresolvable (cyclic or dangling) chain. - TEST_LOG_SUMMARY("FrontierTable::improveChain refused: unverifiable " - "parent chain tag=%lld parent_tag=%lld depth=%u " - "walk_stopped_at=%lld", - (long long)tag, (long long)parent_tag, depth, - (long long)cur); - return false; - } - } - - int idx = (int)(tag - 1); - - _table_lock.lock(); - bool improved = false; - if (idx < _table_size && depth > _table[idx].depth) { - _table[idx].parent_tag = parent_tag; - _table[idx].referrer_klass = referrer_klass; - _table[idx].depth = depth; - _table[idx].root_kind = root_kind; - _table[idx].referrer_field_index = referrer_field_index; - _table[idx].edge_kind = edge_kind; - _table[idx].referrer_class_tag = referrer_class_tag; - improved = true; - } - _table_lock.unlock(); - return improved; -} - -bool FrontierTable::reparentToDurableRoot(jlong tag, jlong new_parent_tag, - u32 referrer_klass, - jint referrer_field_index, - u8 edge_kind) { - // See the declaration's own comment (referenceChains.h) for why this exists as a sibling of - // improveChain(): equal-depth depth-1 noise->real re-parenting. - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX || new_parent_tag <= 0 || - new_parent_tag - 1 >= (jlong)INT_MAX || new_parent_tag == tag) { - return false; - } - int idx = (int)(tag - 1); - int new_par_idx = (int)(new_parent_tag - 1); - - _table_lock.lock(); - bool swapped = false; - if (idx < _table_size && _table[idx].depth == 1 && - _table[idx].parent_tag > 0 && _table[idx].parent_tag != new_parent_tag) { - int old_par_idx = (int)(_table[idx].parent_tag - 1); - if (old_par_idx >= 0 && old_par_idx < _table_size && - new_par_idx < _table_size && - _table[new_par_idx].parent_tag == 0 && - _table[new_par_idx].root_kind != 0 && - !isTransientRootKind(_table[new_par_idx].root_kind) && - _table[old_par_idx].parent_tag == 0 && - isTransientRootKind(_table[old_par_idx].root_kind)) { - // New parent is a root-attached DURABLE root (static field, JNI global, thread) and the - // current parent is a root-attached TRANSIENT one - same depth, strictly better retention - // explanation. - _table[idx].parent_tag = new_parent_tag; - _table[idx].referrer_klass = referrer_klass; - _table[idx].referrer_field_index = referrer_field_index; - _table[idx].edge_kind = edge_kind; - swapped = true; - } - } - _table_lock.unlock(); - return swapped; -} - -bool FrontierTable::reconstructChain(jlong target_tag, - std::vector *out_chain, - u8 *out_root_kind, - std::vector *out_edges, - FrontierEntry *out_terminal) { - FrontierEntry entry{}; - if (!lookup(target_tag, &entry)) { - return false; - } - - std::vector chain; - std::vector edges; - jlong tag = target_tag; - u8 root_kind = 0; - int hops = 0; - // Bounded by maxCapacity(): every tag maps to a distinct slot (this table's "tags/slots are never - // reused" invariant, see the class comment above), so a well-formed parent_tag chain can visit at - // most maxCapacity() slots before either reaching parent_tag == 0 or repeating a slot. - for (; hops <= maxCapacity() && tag != 0; hops++) { - if (!lookup(tag, &entry)) { - // parent_tag pointed at a tag that was never inserted - should not happen for a chain built - // entirely within one BFS pass, but do not fabricate a partial chain silently. - TEST_LOG_SUMMARY("FrontierTable::reconstructChain broken chain: " - "target=%lld failed at hop=%d tag=%lld (parent tag never " - "inserted)", - (long long)target_tag, hops, (long long)tag); - return false; - } - chain.push_back(entry.referrer_klass); - if (out_edges != nullptr) { - // edges[i] describes the edge INTO chain[i]: the entry's own recorded edge identity, plus the - // referrer's class tag - the parent entry's own class for interior hops, the declaring class - // for root-attached static edges (FrontierEntry::referrer_class_tag, filled only there, since - // a class-object referrer has no parent entry to read from). - ChainHopEdge hop{}; - hop.field_index = entry.referrer_field_index; - if (entry.parent_tag == 0) { - hop.edge_kind = entry.root_kind; - hop.referrer_class_tag = entry.referrer_class_tag; - } else { - hop.edge_kind = entry.edge_kind; - FrontierEntry parent_entry{}; - hop.referrer_class_tag = - lookup(entry.parent_tag, &parent_entry) ? parent_entry.class_tag : 0; - } - edges.push_back(hop); - } - // Deliberately NOT marking the walked entries EDGE: the EDGE state was write-only "degenerate - // EdgeStore" bookkeeping (nothing ever reads it), while the rotation collectors select EXPANDED - // entries - demoting a resolved path's holders to EDGE made them permanently invisible to - // rotation, so later leak instances behind a changed holder were never re-discovered. - root_kind = entry.root_kind; - tag = entry.parent_tag; - } - if (tag != 0) { - // Ran past the defensive hop bound without reaching a root-attached entry (parent_tag == 0) - a - // corrupted/cyclic chain. - { - jlong dbg = target_tag; - FrontierEntry dbg_e{}; - char pairs[256]; - size_t off = 0; - for (int d = 0; d < 12 && dbg != 0 && off < sizeof(pairs) - 24; d++) { - if (!lookup(dbg, &dbg_e)) { - break; - } - off += (size_t)snprintf(pairs + off, sizeof(pairs) - off, "%lld->%lld ", - (long long)dbg, (long long)dbg_e.parent_tag); - dbg = dbg_e.parent_tag; - } - TEST_LOG_SUMMARY("FrontierTable::reconstructChain hop bound: " - "target=%lld stuck at tag=%lld after %d hops - cyclic or " - "corrupt parent chain; hops: %.*s", - (long long)target_tag, (long long)tag, hops, (int)off, pairs); - } - return false; - } - - *out_chain = std::move(chain); - if (out_edges != nullptr) { - *out_edges = std::move(edges); - } - if (out_root_kind != nullptr) { - // The loop's last iteration is always the root-attached entry (the one whose parent_tag == 0 - // that just ended the loop), so root_kind here is that entry's own FrontierEntry::root_kind. - *out_root_kind = root_kind; - } - if (out_terminal != nullptr) { - // `entry` still holds the loop's last successful lookup - the root-attached entry that ended - // the walk. - *out_terminal = entry; - } - return true; -} - diff --git a/ddprof-lib/src/main/cpp/referenceChainFrontier.h b/ddprof-lib/src/main/cpp/referenceChainFrontier.h deleted file mode 100644 index e744e840e8..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainFrontier.h +++ /dev/null @@ -1,288 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - - -#ifndef _REFERENCECHAINFRONTIER_H -#define _REFERENCECHAINFRONTIER_H - -#include "arch.h" -#include "common.h" -#include "spinLock.h" -#include -#include -#include -#include -#include -#include -#include - -namespace FrontierEntryState { -constexpr u8 FRONTIER = 0; // discovered, not yet expanded by FollowReferences -constexpr u8 EXPANDED = 1; // expanded; children (if any) are in the table -constexpr u8 EDGE = 2; // on a path toward a target sample (EdgeStore) -constexpr u8 ABANDONED = 3; // tag released; entry kept only to avoid reuse -} // namespace FrontierEntryState - -// Search outcome; per-pass truncation does not imply abandonment. -namespace SearchState { -constexpr u8 RUNNING = 0; // at least one more pass may still make progress -constexpr u8 COMPLETED = 1; // reachable graph fully explored within caps -constexpr u8 ABANDONED = 2; // TTL or frontier-size cap forced an incomplete stop -} // namespace SearchState - -// Reason reported when a search is abandoned. -namespace SearchAbandonReason { -constexpr u8 NONE = 0; // not (yet) abandoned -constexpr u8 FRONTIER_CAP = 1; // frontier-size cap hit -constexpr u8 TTL = 2; // wall-clock TTL exceeded with work still pending -// Canary candidate-discovery has made no progress for NO_PROGRESS_PASS_LIMIT consecutive passes. -constexpr u8 CANARY_STUCK = 3; -} // namespace SearchAbandonReason - -// Metadata for one tagged frontier object. -typedef struct FrontierEntry { - jlong parent_tag; // links back to the record that discovered this one - u32 referrer_klass; // StringDictionary id, 0 = unresolved/none - u32 depth; // hop count from the frontier's seed, for the hop cap - u8 state; // one of FrontierEntryState's constants - // The leak tag assigned by LivenessTracker to this specific tracked object, copied from the JVMTI - // tag at admission time. - jlong leak_tag; - // jvmtiHeapReferenceKind of the edge that admitted this entry, but only meaningful when - // parent_tag == 0 (this entry is root-attached) - 0 (no JVMTI_HEAP_REFERENCE_* value is 0) for - // every other entry, since a non-root entry's own referrer edge kind is not what - // reconstructChain()'s callers want to report (they want to label the chain's root, not every - // hop). - u8 root_kind; - // Raw JVMTI class tag of THIS entry's own object, from the shared, process-wide allocator - // (classTagAllocator.h) - NOT referrer_klass above (a classMap dictionary id, which can differ - // for the same class at different times if that dictionary gets compacted/regenerated - see - // LivenessTracker::KlassPopulationEntry::stable_class_tag's own comment for the bug this was - // found fixing). - jlong class_tag; - - // Retention-edge identity of the edge that admitted THIS entry, captured at admission time for - // the same cannot-replay-the-callback reason as class_tag above: it lets the emitted - // datadog.ReferenceChain name the field each hop is retained through, turning the bare class list - // into a readable path ("LeakHolder.SINK -> HashMap.table -> Entry.value") - which is the - // diagnostic point of the whole feature. - jint referrer_field_index; - - // jvmtiHeapReferenceKind of the admitting edge for an INTERIOR hop (an entry with parent_tag != 0 - // - "this object was reached from its parent via this kind of edge"). - u8 edge_kind; - - // Referrer's class tag when the referrer is a CLASS OBJECT rather than a frontier entry (a - // root-attached static-field admission - the referrer is the declaring class, parent_tag == 0 so - // there is no parent entry to read a class from). - jlong referrer_class_tag; -} FrontierEntry; - -// Per-hop retention-edge identity collected by reconstructChain() alongside the class chain: -// everything needed at emission to label HOW chain[i] is retained (the field of chain[i+1] pointing -// at it, or the root edge for the last hop). -typedef struct ChainHopEdge { - // FrontierEntry::referrer_field_index/edge_kind of the entry for chain[i] - jint field_index; // -1 = not a field/static-field edge - u8 edge_kind; // admitting edge kind (root hops: root_kind) - // The referrer's raw class tag: the PARENT entry's class_tag for interior hops, - // FrontierEntry::referrer_class_tag for root-attached hops. - jlong referrer_class_tag; -} ChainHopEdge; - -// Higher values identify longer-lived root kinds. -inline int rootKindDurability(u8 root_kind) { - switch (root_kind) { - case JVMTI_HEAP_REFERENCE_STATIC_FIELD: - case JVMTI_HEAP_REFERENCE_SYSTEM_CLASS: - return 3; - case JVMTI_HEAP_REFERENCE_JNI_GLOBAL: - return 2; - case JVMTI_HEAP_REFERENCE_MONITOR: - case JVMTI_HEAP_REFERENCE_STACK_LOCAL: - case JVMTI_HEAP_REFERENCE_JNI_LOCAL: - case JVMTI_HEAP_REFERENCE_THREAD: - case JVMTI_HEAP_REFERENCE_OTHER: - return 1; - default: - return 0; // root_kind's own "not set"/non-root-attached value - } -} - -// Stack and JNI locals are transient roots. -inline bool isTransientRootKind(u8 root_kind) { - return root_kind == JVMTI_HEAP_REFERENCE_STACK_LOCAL || - root_kind == JVMTI_HEAP_REFERENCE_JNI_LOCAL; -} - -// Tag-indexed slot table storing FrontierEntry metadata, modeled on LivenessTracker's TrackingEntry -// table (livenessTracker.h:21-30): CAS-safe doubling resize under a signal-safe SpinLock -// (spinLock.h), reusing its shared/exclusive split so reads (lookup) never race a resize. -class alignas(alignof(SpinLock)) FrontierTable { -private: - // Provisional default pending empirical tuning - not benchmark-derived. - static constexpr int INITIAL_TABLE_CAPACITY = 1024; - - // mutable: capacity()/maxCapacity() below are const accessors that still need to take this lock - // to read _table_cap/_table_max_cap safely. - mutable SpinLock _table_lock; - // 1 + highest index ever inserted (informational upper bound for lookup(); never shrinks, since - // tags/slots are never reused). - std::atomic _table_size; - int _table_cap; - int _table_max_cap; - FrontierEntry *_table; - - // Grows _table (doubling) until it holds at least `required_cap` slots or _table_max_cap is - // reached. - bool growLocked(int required_cap); - -public: - // `max_cap` <= 0 disables the table (capacity() stays 0, every insert() reports exhaustion) - - // callers are expected to guard on the config flag before constructing one, but this makes a - // misconfigured cap fail safe rather than crash. - explicit FrontierTable(int max_cap); - ~FrontierTable(); - - FrontierTable(const FrontierTable &) = delete; - FrontierTable &operator=(const FrontierTable &) = delete; - - // Writes (parent_tag, referrer_klass, depth, state) into the slot for `tag` (index = tag - 1), - // growing the table if needed. - bool insert(jlong tag, jlong parent_tag, u32 referrer_klass, u32 depth, - u8 state = FrontierEntryState::FRONTIER, u8 root_kind = 0, - jlong class_tag = 0, - jint referrer_field_index = -1, u8 edge_kind = 0, - jlong referrer_class_tag = 0); - - // Reads the slot for `tag` into *out. Returns false (leaving *out untouched) if `tag` is not - // positive or has never been inserted. - bool lookup(jlong tag, FrontierEntry *out); - - // Runs `fn(this)` with the shared lock held for the whole call, for a caller that needs to look - // up many tags back to back (e.g. the rotation collectors' O(size()) sweeps in - // referenceChains.cpp) under ONE lock acquisition, instead of paying SpinLock's lock/unlock cost - // on every single lookup() call. - template void withSharedLock(Fn &&fn) const { - SharedLockGuard guard(&_table_lock); - fn(this); - } - - // Same as lookup() above, but assumes the caller already holds the shared lock via - // withSharedLock() below. - bool lookupLocked(jlong tag, FrontierEntry *out) const; - - // Marks metadata abandoned; the caller must clear the JVMTI tag. - void clear(jlong tag); - - // Marks the slot as part of a resolved path. - void markEdge(jlong tag); - - // Marks the slot after all outgoing edges have been visited. - void markExpanded(jlong tag); - - // Updates only the recorded root kind. - void updateRootKind(jlong tag, u8 root_kind); - - // Set the leak tag on a frontier entry (the JVMTI tag assigned by LivenessTracker to this - // specific tracked leaking object). - void setLeakTag(jlong tag, jlong leak_tag) { - if (tag <= 0 || tag - 1 >= (jlong)INT_MAX) { - return; - } - int idx = (int)(tag - 1); - _table_lock.lock(); - if (idx < _table_size) { - _table[idx].leak_tag = leak_tag; - } - _table_lock.unlock(); - } - - // Replace a shallow root-attached entry (parent_tag == 0, depth == 0) with a deeper - // chain-attached entry when the object is reached via a longer path. - bool improveChain(jlong tag, jlong parent_tag, u32 referrer_klass, - u32 depth, u8 root_kind, jint referrer_field_index = -1, - u8 edge_kind = 0, jlong referrer_class_tag = 0); - - // Equal-depth re-parenting, the one case improveChain() above cannot express: a depth-1 entry - // whose current parent is a TRANSIENT root (stack local / JNI local - a momentarily-live frame) - // is re-parented to a DURABLE root-attached parent (static field, JNI global, thread) when one is - // seen admitting the same object at the same depth. - bool reparentToDurableRoot(jlong tag, jlong new_parent_tag, - u32 referrer_klass, - jint referrer_field_index = -1, u8 edge_kind = 0); - - // Walks parent_tag links starting at `target_tag` back to a root-attached entry (parent_tag == - // 0), appending each visited entry's referrer_klass to *out_chain in leaf-to-root order. - bool reconstructChain(jlong target_tag, std::vector *out_chain, - u8 *out_root_kind = nullptr, - std::vector *out_edges = nullptr, - FrontierEntry *out_terminal = nullptr); - - // Search restart (ReferenceChainTracker::restartSearch(), this class's own header comment): marks - // every slot unoccupied again without releasing _table's allocation - a new search's nextTag() - // sequence restarts at 1, reusing these same slot indices, so lookup()/insert() must not read - // back the previous search's now-irrelevant entries for them. - void resetForRestart() { - _table_lock.lock(); - _table_size.store(0, std::memory_order_relaxed); - _table_lock.unlock(); - } - - // Debug-only test seam (ReferenceChainTracker::resetSearchStateForTest()). - void resetCapacityForTest(int max_cap); - - // _table_cap/_table_max_cap are plain ints, not atomics like _table_size, and - // resetCapacityForTest() (debug-only test seam, see its own comment) rewrites both under - // _table_lock after freeing/reallocating _table. - int capacity() const { - _table_lock.lock(); - int cap = _table_cap; - _table_lock.unlock(); - return cap; - } - int maxCapacity() const { - _table_lock.lock(); - int max_cap = _table_max_cap; - _table_lock.unlock(); - return max_cap; - } - - // Current upper bound on assigned slots (mirrors _table_size's own comment: "1 + highest index - // ever inserted"). - int size() const { return _table_size.load(std::memory_order_relaxed); } -}; - -// Tag-indexed table mapping a *class* tag (see ReferenceChainTracker::nextClassTag() - always -// negative, a namespace disjoint from the positive FrontierTable object tags above so a raw tag -// value alone always tells the heap-walk callback which table it belongs to) to the -// StringDictionary id of that class's resolved name (Profiler::classMap(), the same interning table -// LivenessTracker uses via Profiler::lookupClass(), livenessTracker.cpp). -class ClassTagTable { -private: - std::unordered_map _table; - -public: - void insert(jlong class_tag, u32 dict_id) { _table[class_tag] = dict_id; } - - // Returns the StringDictionary id for `class_tag`, or 0 if it was never inserted (0 is - // StringDictionary's own "no entry" sentinel too, so this composes with - // FrontierEntry::referrer_klass's documented 0 = unresolved/none convention without a separate - // "found" out-parameter). - u32 resolve(jlong class_tag) const { - auto it = _table.find(class_tag); - return it != _table.end() ? it->second : 0; - } - - size_t size() const { return _table.size(); } - - // Drops every cached class_tag -> dict_id mapping - used when the underlying StringDictionary - // itself was reset (see ReferenceChainTracker::_last_class_map_generation's comment) and every id - // here now points at a namespace that no longer exists. - void clear() { _table.clear(); } -}; - - -#endif // _REFERENCECHAINFRONTIER_H diff --git a/ddprof-lib/src/main/cpp/referenceChainInternal.h b/ddprof-lib/src/main/cpp/referenceChainInternal.h deleted file mode 100644 index 721b5683d4..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainInternal.h +++ /dev/null @@ -1,84 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - - -#ifndef _REFERENCECHAININTERNAL_H -#define _REFERENCECHAININTERNAL_H - -#include "arch.h" -#include -#include - -class FrontierTable; -class ReferenceChainTracker; - -extern thread_local bool t_inGCCallback; - -struct ReferenceChainPassContext { - ReferenceChainTracker *tracker; - FrontierTable *frontier; - int hop_cap; - int budget; - int edges_admitted; - bool truncated; - - // Set only when `truncated` became true because frontier->insert() itself reported capacity - // exhaustion, as opposed to edges_admitted reaching budget. - bool frontier_cap_hit; - - // ARRAY-HOLDER BATCHING: when non-null, expandFrontier() is driving a one-hop expansion of a - // batch of boundary objects passed to a single FollowReferences(initial_object=holder_array) - // call. - std::unordered_set *batch_tags = nullptr; - - // Rolling resume cursor for expandFrontier(): tracks the tag of the last batch entry that - // FollowReferences visited (the callback at the batch_tags descent-gate updates this). - jlong _last_visited_batch_tag = 0; - - // Batch entries the callback finished visiting before the truncation (set only by - // heapReferenceCallback()'s batch_tags descent gate). - std::unordered_set *_completed_batch_tags = nullptr; - - // Set only by admitStaticFieldRoots(): the seed holder array for that sweep holds loaded-class - // objects (negative-tagged by resolveLoadedClasses(), see the *tag_ptr < 0 branch below), and the - // whole point of the sweep is to walk past that holder->class edge into each class's own outgoing - // references - chiefly STATIC_FIELD - which the *tag_ptr < 0 check would otherwise stop cold - // before FollowReferences ever gets to report them. - bool static_field_seed = false; - - // PER-CLASS NON-STATIC QUOTA (admitStaticFieldRoots() only). - jlong _seed_class_tag = 0; // negative tag of the class currently - // being descended (0 before the first class edge is seen) - int _class_other_admitted = 0; // non-STATIC_FIELD edges admitted for - // the current class this lap - int _class_other_cap = 0; // per-class cap; 0 disables the quota - // (admit all) when not in seed sweep - // Number of distinct classes entered so far in this chunk's descent (incremented on each - // class-boundary tag change). - int _classes_in_chunk_visited = 0; - - // Amortizes tracker->_pass_deadline_ns's OS::nanotime() check (heapReference - // Callback()/heapRootCallback() run once per visited edge/root - checking wall-clock on literally - // every call would add real overhead on a large heap) - checked only every 4096th call, local to - // this ctx so each of runPassManualWalk()'s several sub-calls (root enum, static-field sweep, - // expandFrontier(), rotation) starts its own count. - int deadline_check_counter = 0; - - // True while expandFrontier() is walking a batch drawn from _priority_expand (a - // rotation-selected, already-EXPANDED parent) rather than the ordinary _pending_expand backlog - - // see _priority_expand's own comment. - bool admit_priority = false; - - // DESCEND-WALK controls (descendFromAnchor()'s calls only; null/0 everywhere else, so every gate - // below is a no-op for the ordinary walk phases): _no_descend_class_tags: exact class tags to - // neither admit nor descend into for the duration of this walk. - static constexpr int NO_DESCEND_CLASS_CAP = 8; - jlong _no_descend_class_tags[NO_DESCEND_CLASS_CAP] = {0}; - int _no_descend_class_tag_count = 0; - jlong _descent_anchor_tag = 0; - jlong _anchor_descend_class_tag = 0; -}; - -#endif // _REFERENCECHAININTERNAL_H diff --git a/ddprof-lib/src/main/cpp/referenceChainLabels.cpp b/ddprof-lib/src/main/cpp/referenceChainLabels.cpp deleted file mode 100644 index 2bf17d9770..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainLabels.cpp +++ /dev/null @@ -1,328 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -#include "referenceChains.h" -#include "referenceChainInternal.h" -#include "common.h" -#include "counters.h" -#include "jniHelper.h" -#include "jvmThread.h" -#include "livenessTracker.h" -#include "log.h" -#include "objectSampler.h" -#include "os.h" -#include "profiler.h" -#include "rcDebugLevel.h" -#include "tsc.h" -#include "vmEntry.h" -#include -#include -#include -#include -#include -#include -#include -#include - -// Retention-edge labels: naming the field each chain hop is retained -namespace { - -// Fallback label for a hop whose edge is not a field reference (or whose field ordinal could not be -// decoded) - the edge KIND, never a fabricated name. -const char *hopEdgeKindLabel(u8 kind) { - switch (kind) { - case JVMTI_HEAP_REFERENCE_CLASS: - return "class"; - case JVMTI_HEAP_REFERENCE_FIELD: - return "field"; - case JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT: - return "element"; - case JVMTI_HEAP_REFERENCE_CLASS_LOADER: - return "class_loader"; - case JVMTI_HEAP_REFERENCE_SIGNERS: - return "signers"; - case JVMTI_HEAP_REFERENCE_PROTECTION_DOMAIN: - return "protection_domain"; - case JVMTI_HEAP_REFERENCE_INTERFACE: - return "interface"; - case JVMTI_HEAP_REFERENCE_STATIC_FIELD: - return "static_field"; - case JVMTI_HEAP_REFERENCE_CONSTANT_POOL: - return "constant_pool"; - case JVMTI_HEAP_REFERENCE_SUPERCLASS: - return "superclass"; - case JVMTI_HEAP_REFERENCE_JNI_GLOBAL: - return "jni_global"; - case JVMTI_HEAP_REFERENCE_SYSTEM_CLASS: - return "system_class"; - case JVMTI_HEAP_REFERENCE_MONITOR: - return "monitor"; - case JVMTI_HEAP_REFERENCE_STACK_LOCAL: - return "stack_local"; - case JVMTI_HEAP_REFERENCE_JNI_LOCAL: - return "jni_local"; - case JVMTI_HEAP_REFERENCE_THREAD: - return "thread"; - case JVMTI_HEAP_REFERENCE_OTHER: - return "other"; - default: - return "unknown"; - } -} - -// Appends the own-declared field names of `cls` to *out, in GetClassFields() order. -bool appendClassFieldNames(jvmtiEnv *jvmti, JNIEnv *jni, jclass cls, - std::vector *out) { - jint count = 0; - jfieldID *fields = nullptr; - if (jvmti->GetClassFields(cls, &count, &fields) != JVMTI_ERROR_NONE) { - return false; - } - bool ok = true; - for (jint i = 0; i < count; i++) { - char *name = nullptr; - if (jvmti->GetFieldName(cls, fields[i], &name, nullptr, nullptr) != - JVMTI_ERROR_NONE || - name == nullptr) { - ok = false; - break; - } - out->emplace_back(name); - jvmti->Deallocate((unsigned char *)name); - } - jvmti->Deallocate((unsigned char *)fields); - return ok; -} - -// Counts `iface`'s own declared fields plus every transitive superinterface's, each interface -// counted exactly once (keyed on the shared `seen` set insert - an interface's own fields go in -// only when its tag is newly seen, so a diamond's shared parent contributes once no matter how many -// branches reach it). -jlong interfaceSubtreeFieldCount(jvmtiEnv *jvmti, JNIEnv *jni, jclass iface, - std::unordered_set *seen) { - jlong tag = 0; - if (jvmti->GetTag(iface, &tag) != JVMTI_ERROR_NONE || tag == 0) { - // Untagged interface: cannot dedupe reliably - fail the whole decode rather than risk double - // counting. - return -1; - } - if (!seen->insert(tag).second) { - return 0; // already counted this interface (a shared subinterface) - } - jlong total = 0; - jint field_count = 0; - jfieldID *fields = nullptr; - if (jvmti->GetClassFields(iface, &field_count, &fields) == - JVMTI_ERROR_NONE) { - // This interface's own fields count toward any implementor's ordinal base - matching the spec's - // "count of the fields in all the interfaces implemented by C" (jvmtiHeapReferenceInfoField). - total += field_count; - jvmti->Deallocate((unsigned char *)fields); - } - jint iface_count = 0; - jclass *supers = nullptr; - if (jvmti->GetImplementedInterfaces(iface, &iface_count, &supers) != - JVMTI_ERROR_NONE) { - return -1; - } - bool ok = true; - for (jint i = 0; i < iface_count; i++) { - if (supers[i] == nullptr) { - continue; - } - jlong sub = interfaceSubtreeFieldCount(jvmti, jni, supers[i], seen); - jni->DeleteLocalRef(supers[i]); - if (sub < 0) { - ok = false; - } else if (ok) { - total += sub; - } - } - jvmti->Deallocate((unsigned char *)supers); - return ok ? total : -1; -} - -// Sums the field counts of every interface transitively implemented/extended by `cls`, each -// interface counted exactly once (see interfaceSubtreeFieldCount() above - the own-field add lives -// behind the shared seen-set insert, so an interface diamond no longer double-counts its shared -// parent). -jlong interfaceFieldCount(jvmtiEnv *jvmti, JNIEnv *jni, jclass cls, - std::unordered_set *seen) { - jint iface_count = 0; - jclass *ifaces = nullptr; - if (jvmti->GetImplementedInterfaces(cls, &iface_count, &ifaces) != - JVMTI_ERROR_NONE) { - return -1; - } - jlong total = 0; - bool ok = true; - for (jint i = 0; i < iface_count; i++) { - if (ifaces[i] == nullptr) { - continue; - } - jlong sub = interfaceSubtreeFieldCount(jvmti, jni, ifaces[i], seen); - jni->DeleteLocalRef(ifaces[i]); - if (sub < 0) { - ok = false; - } else if (ok) { - total += sub; - } - } - jvmti->Deallocate((unsigned char *)ifaces); - return ok ? total : -1; -} - -} // namespace - -const ReferenceChainTracker::HopLabelClass * -ReferenceChainTracker::hopLabelClassFor(jvmtiEnv *jvmti, JNIEnv *jni, - jlong class_tag) { - auto it = _hop_label_cache.find(class_tag); - if (it != _hop_label_cache.end()) { - return &it->second; - } - // Bounded: chains reference few distinct referrer classes; a wholesale clear at the cap (rather - // than LRU eviction) keeps this O(1) and is correct because the cache is purely derived state - - // any cleared entry is transparently rebuilt on its next hop. - if (_hop_label_cache.size() >= HOP_LABEL_CLASS_CACHE_CAP) { - _hop_label_cache.clear(); - } - HopLabelClass entry{}; - entry.class_tag = class_tag; - entry.decode_failed = true; // until proven otherwise - do { - // GetSuperclass is a JNI (not JVMTI) function - modern JVMTI dropped it (the spec delivers - // superclass references via heap callbacks, jvmti.xml's JVMTI_HEAP_REFERENCE_SUPERCLASS note); - // the rest are JVMTI slots. - if (jvmti->functions->GetObjectsWithTags == nullptr || - jvmti->functions->IsInterface == nullptr || - jvmti->functions->GetImplementedInterfaces == nullptr || - jvmti->functions->GetClassFields == nullptr || - jvmti->functions->GetFieldName == nullptr || - jni->functions->GetSuperclass == nullptr) { - break; - } - // Resolve the class object from its raw tag (negative - the shared allocator's class tags; - // GetObjectsWithTags accepts any tag value). - jint count = 0; - jobject *objects = nullptr; - jlong *tags = nullptr; - if (jvmti->GetObjectsWithTags(1, &class_tag, &count, &objects, &tags) != - JVMTI_ERROR_NONE || - count != 1 || objects == nullptr || objects[0] == nullptr) { - // Both result arrays are JVMTI-allocated on success and must be Deallocate()d by the caller - - // same contract as every other GetObjectsWithTags() call site in this file. - if (objects != nullptr) { - jvmti->Deallocate((unsigned char *)objects); - } - if (tags != nullptr) { - jvmti->Deallocate((unsigned char *)tags); - } - break; - } - jclass cls = (jclass)objects[0]; - // The arrays were only needed to obtain the class object - the jobject handle stays valid on - // its own - so release them before the (multiple, break-exited) decode branches below, which - // otherwise all leak them. - jvmti->Deallocate((unsigned char *)objects); - jvmti->Deallocate((unsigned char *)tags); - jboolean is_interface = JNI_FALSE; - std::vector names; - bool ok = false; - if (jvmti->IsInterface(cls, &is_interface) == JVMTI_ERROR_NONE) { - if (is_interface) { - // The spec's INTERFACE branch: base = fields of all superinterfaces of I, then I's own - // fields (jvmtiHeapReferenceInfoField). - std::unordered_set seen; - jlong base = interfaceFieldCount(jvmti, jni, cls, &seen); - if (base >= 0) { - names.resize((size_t)base); // positioned but unnamed: ordinal [0, - // base) is interface fields, only reachable through an - // interface branch decode of a superinterface - ok = appendClassFieldNames(jvmti, jni, cls, &names); - } - } else { - // The spec's CLASS branch: base = fields of all interfaces implemented by C, then the - // superclass chain root-first (java.lang.Object's fields first, C's own last), each class's - // fields in GetClassFields() order. - std::unordered_set seen; - jlong base = interfaceFieldCount(jvmti, jni, cls, &seen); - if (base >= 0) { - names.resize((size_t)base); - // GetSuperclass walks UP, so gather then append in reverse (root first). - jclass supers[128]; - int depth = 0; - jclass k = cls; - while (k != nullptr && - depth < (int)(sizeof(supers) / sizeof(supers[0]))) { - supers[depth++] = k; - k = jni->GetSuperclass(k); - } - ok = (k == nullptr); // deeper than 128 classes: fail rather than - // misname - for (int i = depth - 1; ok && i >= 0; i--) { - ok = appendClassFieldNames(jvmti, jni, supers[i], &names); - } - for (int i = 0; i < depth; i++) { - jni->DeleteLocalRef(supers[i]); - } - // supers[0] IS cls - the loop above already deleted it. Null it so the shared cleanup - // below does not delete the same local ref a second time (checked JNI reports an invalid - // local ref and aborts). - cls = nullptr; - } - } - } - if (cls != nullptr) { - jni->DeleteLocalRef(cls); - } - if (!ok) { - break; - } - entry.field_names = std::move(names); - entry.decode_failed = false; - } while (false); - auto inserted = _hop_label_cache.emplace(class_tag, std::move(entry)); - return &inserted.first->second; -} - -void ReferenceChainTracker::resolveHopEdgeLabel(jvmtiEnv *jvmti, JNIEnv *jni, - ChainHopEdge edge, char *out, - size_t out_cap) { - if (jvmti != nullptr && jni != nullptr && - (edge.edge_kind == JVMTI_HEAP_REFERENCE_FIELD || - edge.edge_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) && - edge.field_index >= 0 && edge.referrer_class_tag != 0 && out_cap > 0) { - const HopLabelClass *labels = - hopLabelClassFor(jvmti, jni, edge.referrer_class_tag); - if (labels != nullptr && !labels->decode_failed && - (size_t)edge.field_index < labels->field_names.size()) { - const std::string &name = - labels->field_names[(size_t)edge.field_index]; - if (!name.empty()) { - size_t n = name.size() < out_cap - 1 ? name.size() : out_cap - 1; - memcpy(out, name.data(), n); - out[n] = '\0'; - return; - } - // Empty positioned slot (an interface-field ordinal below the class's own base) - fall - // through to the kind label. - } - } - if (out_cap > 0) { - snprintf(out, out_cap, "%s", hopEdgeKindLabel(edge.edge_kind)); - } -} - -void ReferenceChainTracker::fillHopEdgeLabels( - jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &edges, - std::vector *out) { - char label[MAX_HOP_EDGE_LABEL + 1]; - for (size_t i = 0; i < edges.size() && i < out->size(); i++) { - resolveHopEdgeLabel(jvmti, jni, edges[i], label, sizeof(label)); - (*out)[i].edge_label = label; - } -} - diff --git a/ddprof-lib/src/main/cpp/referenceChainTraversal.cpp b/ddprof-lib/src/main/cpp/referenceChainTraversal.cpp deleted file mode 100644 index 6c142029ab..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainTraversal.cpp +++ /dev/null @@ -1,1301 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -#include "referenceChains.h" -#include "referenceChainInternal.h" -#include "common.h" -#include "counters.h" -#include "jniHelper.h" -#include "jvmThread.h" -#include "livenessTracker.h" -#include "log.h" -#include "objectSampler.h" -#include "os.h" -#include "profiler.h" -#include "rcDebugLevel.h" -#include "tsc.h" -#include "vmEntry.h" -#include -#include -#include -#include -#include -#include -#include -#include - -// Manual walk driver - IterateOverReachableObjects root/stack-ref enumeration - -namespace { -// jvmtiHeapRootKind (IterateOverReachableObjects's root/stack-ref callbacks, ordinals 1-7) and -// jvmtiHeapReferenceKind (FrontierEntry::root_kind's own type, FollowReferences' callback, ordinals -// 8/21-27) are different, disjoint enums per the real jvmti.h - storing a raw jvmtiHeapRootKind -// value into root_kind unmodified would make flightRecorder.cpp's rootKindName() report "unknown" -// for every root-callback-attributed chain. -u8 translateHeapRootKind(jvmtiHeapRootKind root_kind) { - switch (root_kind) { - case JVMTI_HEAP_ROOT_JNI_GLOBAL: - return (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL; - case JVMTI_HEAP_ROOT_SYSTEM_CLASS: - return (u8)JVMTI_HEAP_REFERENCE_SYSTEM_CLASS; - case JVMTI_HEAP_ROOT_MONITOR: - return (u8)JVMTI_HEAP_REFERENCE_MONITOR; - case JVMTI_HEAP_ROOT_STACK_LOCAL: - return (u8)JVMTI_HEAP_REFERENCE_STACK_LOCAL; - case JVMTI_HEAP_ROOT_JNI_LOCAL: - return (u8)JVMTI_HEAP_REFERENCE_JNI_LOCAL; - case JVMTI_HEAP_ROOT_THREAD: - return (u8)JVMTI_HEAP_REFERENCE_THREAD; - case JVMTI_HEAP_ROOT_OTHER: - default: - return (u8)JVMTI_HEAP_REFERENCE_OTHER; - } -} - -} // namespace - -jvmtiIterationControl JNICALL ReferenceChainTracker::heapRootCallback( - jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, - void *user_data) { - ReferenceChainPassContext *ctx = (ReferenceChainPassContext *)user_data; - if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { - ctx->truncated = true; - return JVMTI_ITERATION_ABORT; - } - if (ctx->truncated) { - return JVMTI_ITERATION_ABORT; - } - - u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); - u8 translated_root_kind = translateHeapRootKind(root_kind); - AdmitResult result = ctx->tracker->admitObject( - ctx->frontier, ctx->hop_cap, ctx->budget, &ctx->edges_admitted, tag_ptr, - /*parent_tag=*/0, referrer_klass, /*depth=*/0, translated_root_kind, - class_tag); - switch (result) { - case AdmitResult::BUDGET_EXHAUSTED: - ctx->truncated = true; - return JVMTI_ITERATION_ABORT; - case AdmitResult::FRONTIER_CAP_HIT: - ctx->truncated = true; - ctx->frontier_cap_hit = true; - return JVMTI_ITERATION_ABORT; - case AdmitResult::ALREADY_ADMITTED: - if (isLeakTag(*tag_ptr)) { - // Leak-tagged object met as a DIRECT heap root (JNI global, stack local, ...): convert the - // leak tag exactly like heapReferenceCallback()'s interception branch so a chain can be built - // when the direct root is the only retention path. - jlong leak_tag = *tag_ptr; - jlong frontier_tag = ctx->tracker->nextTag(); - if (ctx->frontier->insert(frontier_tag, /*parent_tag=*/0, referrer_klass, - /*depth=*/0, FrontierEntryState::FRONTIER, - translated_root_kind, class_tag)) { - ctx->frontier->setLeakTag(frontier_tag, leak_tag); - *tag_ptr = frontier_tag; - ctx->edges_admitted++; - ctx->tracker->trackLeakAccumulation(ctx->frontier, class_tag, 0, - frontier_tag); - // Index maintenance, mirroring the edge path's interception branch: a leak-tagged root - // attached by a durable root edge is the highest-priority anchor tier. - if (translated_root_kind == (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || - translated_root_kind == (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) { - ctx->tracker->addToStaticAnchorIndex(frontier_tag, class_tag, - translated_root_kind); - } - if (ctx->tracker->_candidate_count > 0) { - u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); - ctx->tracker->recordDiscoveredInstance(klass_id, frontier_tag, true); - } - // Queue for expandFrontier() - the plain backlog lane, mirroring admitObject()'s - // non-priority push tail. - ctx->tracker->_pending_expand.push_back(frontier_tag); - } else { - // Frontier cap hit - same outcome as admitObject()'s own failure. - ctx->truncated = true; - ctx->frontier_cap_hit = true; - return JVMTI_ITERATION_ABORT; - } - break; - } - // Prefer the more durable root when an object is rediscovered. - ctx->tracker->maybeUpgradeRootAttachedRootKind(ctx->frontier, *tag_ptr, - translated_root_kind); - break; - default: - break; - } - return JVMTI_ITERATION_CONTINUE; -} - -jvmtiIterationControl JNICALL ReferenceChainTracker::stackRefCallback( - jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, - jlong thread_tag, jint depth, jmethodID method, jint slot, - void *user_data) { - // Stack-local/JNI-local roots carry thread/frame/slot detail JVMTI reports via this callback's - // richer shape, but FrontierEntry has nowhere to record it (depth/method/slot are not part of the - // record) - admission is otherwise identical to heapRootCallback() above, so this just forwards. - return heapRootCallback(root_kind, class_tag, size, tag_ptr, user_data); -} - -void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, - bool run_root_enum, - int root_enum_budget, - int expand_budget, - int *edges_admitted, - bool *truncated, - bool *frontier_cap_hit, - u64 *safepoint_ticks) { - assert(!t_inGCCallback && - "IterateOverReachableObjects/FollowReferences are JVMTI " - "Heap-category calls and must not be made from " - "GarbageCollectionStart/Finish"); - - // Safe point to delete the global refs of threads that ended since the last drain: this runs on - // the BFS thread before any walk phase, and refs erased from _thread_objects - // (unregisterThreadObject()) can no longer be copied out by walkCandidateThreadLocals(), so no - // walk holds them. - releaseEndedThreadRefs(jni); - - *safepoint_ticks = 0; - - // Shared wall-clock ceiling for this whole call's static-field sweep, expandFrontier(), and - // rotation sub-calls below (see _pass_deadline_ns's own comment) - deliberately NOT applied to - // root/stack-ref enumeration itself, which is instead cadence-gated by run_root_enum/ - // ROOT_ENUM_MIN_INTERVAL_NS. - _pass_deadline_ns = _effective_pause_target_ms > 0 - ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL - : 0; - - *edges_admitted = 0; - *truncated = false; - *frontier_cap_hit = false; - - // Reserve a slice for rotation up front, across all three tiers (see - // ROOT_KIND_ROTATION_BUDGET/LEAK_ACCUMULATION_ROTATION_BUDGET/ STALE_EXPANDED_ROTATION_BUDGET's - // own comments) so rotation still gets to run this pass even when ordinary work below spends - // everything else and truncates. - int rotation_reserved_budget = std::min( - expand_budget / 2, ROOT_KIND_ROTATION_BUDGET + - LEAK_ACCUMULATION_ROTATION_BUDGET + - STALE_EXPANDED_ROTATION_BUDGET); - int budget = expand_budget - rotation_reserved_budget; - - // Root/stack-ref enumeration alone (unlike a root-seeded FollowReferences call on the fallback - // path) never discovers a root's own transitive children - IterateOverReachableObjects's - // root/stack-ref callbacks are given no oop, only a tag_ptr (see heapRootCallback()'s own - // comment) - so even when it runs this pass, the expandFrontier() call below is still needed to - // make any further progress. - if (run_root_enum) { - ReferenceChainPassContext ctx; - ctx.tracker = this; - ctx.frontier = _frontier; - ctx.hop_cap = _hop_cap; - ctx.budget = root_enum_budget; - ctx.edges_admitted = 0; - ctx.truncated = false; - ctx.frontier_cap_hit = false; - - u64 root_enum_start_ticks = TSC::ticks(); - jvmtiError root_err = jvmti->IterateOverReachableObjects( - heapRootCallback, stackRefCallback, /*object_ref_callback=*/nullptr, - &ctx); - *safepoint_ticks += TSC::ticks() - root_enum_start_ticks; - - // expand_budget is spent independently of root_enum_budget below (see - // ROOT_ENUM_MIN_INTERVAL_NS's own comment) - ctx.edges_admitted is written straight into - // *edges_admitted so the static-field/expand/ rotation budget math below is never shrunk by - // whatever root enumeration admitted. - *edges_admitted = ctx.edges_admitted; - _last_root_enum_ns = OS::nanotime(); - - if (root_err != JVMTI_ERROR_NONE) { - *truncated = true; - *frontier_cap_hit = false; - _root_enum_truncated_last_time = false; - return; - } - if (ctx.truncated) { - *truncated = true; - *frontier_cap_hit = ctx.frontier_cap_hit; - // Only a budget-exhausted truncation (not a frontier-cap-hit, which abandons the search - // outright) is grounds to retry root enumeration on the very next pass - see - // _root_enum_truncated_last_time's own comment. - _root_enum_truncated_last_time = !ctx.frontier_cap_hit; - return; - } - _root_enum_truncated_last_time = false; - } - - int expand_phase_edges_admitted = 0; - - // Candidate-scoped reach, prong 1: descend-walk the current candidates' qualifying threads' - // ThreadLocalMap subgraphs BEFORE any breadth-first work this pass - reaching the tagged - // instances under a thread-retained holder must not queue behind the ordinary backlog (see - // walkCandidateThreadLocals()'s own comment). - if (_candidate_count > 0) { - int thread_walk_edges_admitted = 0; - bool thread_walk_truncated = false; - bool thread_walk_frontier_cap_hit = false; - // Give the thread walk its own fresh deadline so the root-enum walk above never eats its slice - // (per-sub-op reset rationale, see expand below). - _pass_deadline_ns = _effective_pause_target_ms > 0 - ? OS::nanotime() + - (u64)_effective_pause_target_ms * 1000000ULL - : 0; - walkCandidateThreadLocals(jvmti, jni, budget, &thread_walk_edges_admitted, - &thread_walk_truncated, - &thread_walk_frontier_cap_hit, safepoint_ticks); - expand_phase_edges_admitted += thread_walk_edges_admitted; - *edges_admitted += thread_walk_edges_admitted; - if (thread_walk_frontier_cap_hit) { - // Frontier-cap mid-thread-walk is the same search-abandonment grounds as anywhere else - do - // not spend more of this pass's budget. - *truncated = true; - *frontier_cap_hit = true; - return; - } - if (thread_walk_truncated) { - *truncated = true; - } - } - - // Static-field roots (SomeClass.staticField -> obj) are not reachable via - // IterateOverReachableObjects' root/stack-ref callbacks above - see admitStaticFieldRoots()'s own - // comment - so this pass would otherwise never discover an object retained only that way. - TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk static_sweep_gate " - "resolved=%d swept=%d cursor=%d", - _last_resolved_class_count, _last_static_field_class_count, - _static_field_sweep_cursor); - if (_last_resolved_class_count != _last_static_field_class_count) { - int static_field_edges_admitted = 0; - bool static_field_truncated = false; - bool static_field_frontier_cap_hit = false; - bool static_field_cycle_complete = false; - int static_field_budget = std::max(budget - expand_phase_edges_admitted, 0); - admitStaticFieldRoots(jvmti, jni, _hop_cap, static_field_budget, - &static_field_edges_admitted, &static_field_truncated, - &static_field_frontier_cap_hit, - &static_field_cycle_complete, safepoint_ticks); - expand_phase_edges_admitted += static_field_edges_admitted; - *edges_admitted += static_field_edges_admitted; - if (static_field_truncated) { - *truncated = true; - *frontier_cap_hit = static_field_frontier_cap_hit; - if (static_field_frontier_cap_hit) { - // Frontier-size cap hit while admitting static-field roots is the same "grounds to ABANDON - // the whole search" outcome BUDGET_EXHAUSTED/FRONTIER_CAP_HIT handling above gives root - // enumeration - do not spend any more of this pass's budget on the ordinary expansion - // below. - return; - } - } - if (static_field_cycle_complete) { - // The chunk cursor completed a full lap over the loaded-class list with no chunk truncating - // along the way (possibly discovering nothing, if every static field seen was already - // ALREADY_ADMITTED) - remember the class count it covered so a later pass with no new classes - // can skip re-running the sweep entirely. - _last_static_field_class_count = _last_resolved_class_count; - } - } - - int expand_edges_admitted = 0; - bool expand_truncated = false; - bool expand_frontier_cap_hit = false; - int remaining_budget = std::max(budget - expand_phase_edges_admitted, 0); - // Give expand its own fresh deadline so the static-field sweep's FollowReferences calls don't eat - // expand's time. - _pass_deadline_ns = _effective_pause_target_ms > 0 - ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL - : 0; - expandFrontier(jvmti, jni, _hop_cap, remaining_budget, - &expand_edges_admitted, &expand_truncated, - &expand_frontier_cap_hit, safepoint_ticks); - expand_phase_edges_admitted += expand_edges_admitted; - *edges_admitted += expand_edges_admitted; - *truncated = *truncated || expand_truncated; - *frontier_cap_hit = expand_frontier_cap_hit; - TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk expand_phase " - "edges_admitted=%d truncated=%d frontier_cap_hit=%d " - "remaining_budget=%d", - expand_edges_admitted, (int)expand_truncated, - (int)expand_frontier_cap_hit, remaining_budget); - - // Note: unlike a hard truncation during root/stack-ref enumeration or the static-field sweep - // above (which return early - the pass never even reached ordinary expansion), a truncated - // ordinary expansion does NOT skip rotation below: rotation runs on its own reserved slice of - // budget (see rotation_reserved_budget's own comment above) precisely because ordinary expansion - // truncates on nearly every pass under a sustained fast-growing backlog, and that is exactly the - // situation - a mutable field reassigned out from under an already-EXPANDED entry - rotation - // exists to correct. - - // Revisit a bounded subset of expanded entries to observe changed references. - std::vector rotation_tags = - collectStaleRootKindEntriesForRotation(ROOT_KIND_ROTATION_BUDGET); - std::vector leak_accumulation_tags = - collectLeakAccumulationCandidatesForRotation( - LEAK_ACCUMULATION_ROTATION_BUDGET); - // Also re-walk a bounded, rotating subset of EXPANDED entries regardless of root attribution: a - // mutable field reassigned since an object's one-time expansion - e.g. HashMap.table on resize - - // is otherwise never observed again, silently orphaning everything only reachable through the - // field's current value. - std::vector stale_expanded_tags = - collectStaleExpandedEntriesForRotation(STALE_EXPANDED_ROTATION_BUDGET); - // Candidate-scoped reach, prong 2: root-attached static holders are descend-walked directly (see - // collectStaticFieldAnchorsForRotation()/ walkStaticFieldAnchors()'s own comments) - not pushed - // onto the priority lane, so they are independent of the queue tiers above. - reconcileAnchorClassShapes(jvmti, jni); - std::vector static_anchor_tags = - collectStaticFieldAnchorsForRotation(STATIC_ANCHOR_ROTATION_BUDGET); - std::vector static_anchor_fifo_drained; - drainStaticAnchorFifo(STATIC_ANCHOR_FIFO_DRAIN, static_anchor_fifo_drained); - for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { - static_anchor_tags.push_back(at_risk.tag); - } - if (rotation_tags.empty() && leak_accumulation_tags.empty() && - stale_expanded_tags.empty() && static_anchor_tags.empty()) { - return; - } - // rotation_reserved_budget + max(budget - expand_phase_edges_admitted, 0) is exactly - // expand_budget - expand_phase_edges_admitted: budget already IS expand_budget - - // rotation_reserved_budget (above), and expand_phase_edges_ admitted can never exceed budget (the - // static-field sweep and ordinary expandFrontier() calls above are both capped to budget-derived - // slices), so the max() is never actually needed to avoid going negative. - int rotation_budget = expand_budget - expand_phase_edges_admitted; - int rotation_edges_admitted = 0; - bool rotation_truncated = false; - // Give rotation its own fresh deadline, same as expand above. - _pass_deadline_ns = _effective_pause_target_ms > 0 - ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL - : 0; - bool rotation_frontier_cap_hit = false; - // Prong 2 static-anchor descend walks run FIRST inside rotation's slice: they are the - // highest-value rotation work (bounded, targeted, and the only rotation tier that can reach a - // collection-shaped static holder's internals in one pass), and their edges draw down the same - // rotation budget the queue-tier batch below uses - a pass whose anchor walks admit the holder's - // whole internal structure needs less one-hop rotation work, not more. - if (!static_anchor_tags.empty() && rotation_budget > 0) { - int static_anchor_edges_admitted = 0; - bool static_anchor_truncated = false; - bool static_anchor_frontier_cap_hit = false; - std::vector static_anchor_unwalked; - walkStaticFieldAnchors(jvmti, jni, static_anchor_tags, rotation_budget, - &static_anchor_edges_admitted, - &static_anchor_truncated, - &static_anchor_frontier_cap_hit, safepoint_ticks, - &static_anchor_unwalked); - rotation_edges_admitted += static_anchor_edges_admitted; - rotation_budget -= static_anchor_edges_admitted; - *truncated = *truncated || static_anchor_truncated; - // B' requeue: resolved-but-unwalked anchors that came from this pass's FIFO drain go back to - // the FIFO front, order-preserving, so a pass whose budget died mid-batch walks them first next - // pass instead of waiting for the next sweep lap's re-push. - if (!static_anchor_unwalked.empty() && - !static_anchor_fifo_drained.empty()) { - std::vector static_anchor_requeue; - for (jlong tag : static_anchor_unwalked) { - FrontierEntry entry{}; - if (!_frontier->lookup(tag, &entry)) { - continue; - } - for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { - if (tag == at_risk.tag) { - static_anchor_requeue.push_back(at_risk); - break; - } - } - } - if (!static_anchor_requeue.empty()) { - requeueStaticAnchorFifoFront(static_anchor_requeue); - } - } - if (static_anchor_frontier_cap_hit) { - *frontier_cap_hit = true; - return; - } - } - // expandFrontier() SETS (does not add into) its edges output - see its entry - so the anchor - // walks' edges are kept in a separate counter and summed here. - int queue_tier_edges_admitted = 0; - expandFrontier(jvmti, jni, _hop_cap, rotation_budget, - &queue_tier_edges_admitted, &rotation_truncated, - &rotation_frontier_cap_hit, safepoint_ticks); - rotation_edges_admitted += queue_tier_edges_admitted; - *edges_admitted += rotation_edges_admitted; - // OR, not overwrite: the ordinary expand phase above may have already set these to true (real - // truncation/cap-hit left in _pending_expand), and a rotation batch that happens to finish - // cleanly must not erase that - has_pending_frontier (runPass()) and the FRONTIER_CAP abandon - // check both read these as "did any of this pass's sub-phases truncate/cap-hit", not just the - // last one that ran. - *truncated = *truncated || rotation_truncated; - *frontier_cap_hit = *frontier_cap_hit || rotation_frontier_cap_hit; -} - -// Incremental resumption across passes. - -void ReferenceChainTracker::markAllFrontierExpanded() { - while (!_priority_expand.empty()) { - _frontier->markExpanded(_priority_expand.front()); - _priority_expand.pop_front(); - } - _priority_expand_set.clear(); - while (!_pending_expand.empty()) { - _frontier->markExpanded(_pending_expand.front()); - _pending_expand.pop_front(); - } -} - -void ReferenceChainTracker::expandFrontier(jvmtiEnv *jvmti, JNIEnv *jni, - int hop_cap, int budget, - int *edges_admitted, - bool *truncated, - bool *frontier_cap_hit, - u64 *safepoint_ticks) { - assert(!t_inGCCallback && - "GetObjectsWithTags/FollowReferences are JVMTI Heap-category calls " - "and must not be made from GarbageCollectionStart/Finish"); - - ReferenceChainPassContext ctx; - ctx.tracker = this; - ctx.frontier = _frontier; - ctx.hop_cap = hop_cap; - ctx.budget = budget; - ctx.edges_admitted = 0; - ctx.truncated = false; - ctx.frontier_cap_hit = false; - - // ARRAY-HOLDER BATCHING: expand a whole batch of boundary objects with ONE - // FollowReferences(initial_object=holder_array) call per BFS level, instead of one - // FollowReferences PER frontier entry. - std::unordered_set batch_tags; - ctx.batch_tags = &batch_tags; - // Completed-batch-entry tracking for the order-independent truncated-batch resume (see - // ReferenceChainPassContext::_completed_batch_tags) - reset per batch along with the rolling - // cursor. - std::unordered_set completed_batch_tags; - ctx._completed_batch_tags = &completed_batch_tags; - - jvmtiHeapCallbacks callbacks; - memset(&callbacks, 0, sizeof(callbacks)); - callbacks.heap_reference_callback = heapReferenceCallback; - - // java/lang/Object element type for the transient frontier-holder array. - if (jni != nullptr && _cached_object_class == nullptr) { - jclass local = jni->FindClass("java/lang/Object"); - if (!jniExceptionCheck(jni) && local != nullptr) { - _cached_object_class = (jclass)jni->NewGlobalRef(local); - } - if (local != nullptr) { - jni->DeleteLocalRef(local); - } - } - jclass object_class = _cached_object_class; - - bool progress = true; - // FAIR-SHARE DRAIN: alternate batches between _priority_expand and _pending_expand whenever both - // are non-empty (priority still takes the first batch of each call). - while (!ctx.truncated && progress && object_class != nullptr) { - // Wall-clock deadline check per iteration: GetObjectsWithTags runs OUTSIDE any FollowReferences - // callback, so heapReferenceCallback()'s amortized deadline check never sees its cost. - if (_pass_deadline_ns != 0 && OS::nanotime() >= _pass_deadline_ns) { - ctx.truncated = true; - break; - } - progress = false; - - // Alternate lanes (see FAIR-SHARE DRAIN above); priority still goes first so a - // rotation-selected parent's re-discovery keeps its head-of-queue property, but no lane can - // monopolize the drain. - bool from_priority; - if (_priority_expand.empty()) { - from_priority = false; - } else if (_pending_expand.empty()) { - from_priority = true; - } else { - from_priority = _expand_lane_prefer_priority; - _expand_lane_prefer_priority = !_expand_lane_prefer_priority; - } - std::deque &source = - from_priority ? _priority_expand : _pending_expand; - ctx.admit_priority = from_priority; - if (source.empty()) { - break; // nothing pending in either lane - } - - // SELF-CALIBRATING ADAPTIVE BATCH SIZE for GetObjectsWithTags. - size_t gotw_batch_size = - _gotw_batch_size != 0 ? _gotw_batch_size : GOTW_INITIAL_BATCH_SIZE; - size_t batch_size = std::min( - source.size(), - std::min((size_t)std::max(std::min(budget, _budget), 1), - gotw_batch_size)); - std::vector candidate_tags(source.begin(), - source.begin() + batch_size); - - // Resolve this batch's live boundary objects. GetObjectsWithTags iterates the whole tag map, - // but does so under a no-safepoint mutex on this (Java) thread - it is NOT a stop-the-world VM - // operation, unlike the FollowReferences below (jvmtiTagMap.cpp: get_objects_with_tags takes - // Mutex::_no_safepoint_check_flag and calls entry_iterate directly, whereas follow_references - // does VMThread::execute()). - jint resolved_count = 0; - jobject *resolved_objects = nullptr; - jlong *resolved_tags = nullptr; - u64 gotw_start_ns = OS::nanotime(); - jvmtiError resolve_err = jvmti->GetObjectsWithTags( - (jint)candidate_tags.size(), candidate_tags.data(), &resolved_count, - &resolved_objects, &resolved_tags); - u64 gotw_elapsed_ns = OS::nanotime() - gotw_start_ns; - // Self-calibrate (PROPORTIONAL batch control): update the EMA of PER-CALL elapsed time, then - // scale the batch so ONE call fills the remaining wall-clock window. - if (batch_size > 0 && gotw_elapsed_ns > 0) { - if (_gotw_ema_call_ns == 0) { - _gotw_ema_call_ns = gotw_elapsed_ns; - } else { - _gotw_ema_call_ns = _gotw_ema_call_ns * 4 / 5 + gotw_elapsed_ns / 5; - } - u64 now_ns = OS::nanotime(); - u64 window_ns = - gotwWindowNs( - _pass_deadline_ns != 0 && _pass_deadline_ns > now_ns - ? _pass_deadline_ns - now_ns - : 0, - source.size()); - // window_ns / ema_call_ns == how many such calls fit the window; scaling the CURRENT - // calibration batch by that ratio sizes the next call to consume the whole window in one go. - size_t calib_batch = - _gotw_batch_size != 0 ? _gotw_batch_size : GOTW_INITIAL_BATCH_SIZE; - size_t next_batch = (size_t)((u64)calib_batch * window_ns / - std::max(_gotw_ema_call_ns, 1ULL)); - _gotw_batch_size = std::min(std::max(next_batch, GOTW_MIN_BATCH), - GOTW_MAX_BATCH); - } - if (resolve_err != JVMTI_ERROR_NONE) { - ctx.truncated = true; - break; - } - - std::unordered_map live; - for (jint i = 0; i < resolved_count; i++) { - live[resolved_tags[i]] = resolved_objects[i]; - } - - // Build the frontier-holder array from the live boundary objects and record their tags so - // heapReferenceCallback() descends into exactly these (one hop). - batch_tags.clear(); - jobjectArray holder = nullptr; - if (resolved_count > 0) { - jint capacity_err = jni->EnsureLocalCapacity(resolved_count + 16); - if (capacity_err < 0 || jniExceptionCheck(jni)) { - // Could not guarantee local-ref headroom for this batch - treat like any other batch-level - // failure below (JVMTI error / OOM building the holder array): retry this batch on a later - // pass rather than proceeding into NewObjectArray with no capacity guarantee. - ctx.truncated = true; - } else { - holder = jni->NewObjectArray(resolved_count, object_class, nullptr); - if (jniExceptionCheck(jni)) { - // OutOfMemoryError building the holder array (or any other exception NewObjectArray - // raised) left `holder` null; make sure the pending exception does not survive into the - // next JNI call below or the next expandFrontier() invocation on this same long-lived - // BFS-thread JNIEnv (JNI spec: undefined behavior with a pending exception across - // ordinary JNI calls). - holder = nullptr; - } - if (holder != nullptr) { - for (jint i = 0; i < resolved_count; i++) { - jni->SetObjectArrayElement(holder, i, resolved_objects[i]); - if (jniExceptionCheck(jni)) { - // e.g. an array-store-class failure. Abort building this batch's holder rather than - // handing a partially-populated array (with a just-cleared pending exception) to - // FollowReferences. - ctx.truncated = true; - break; - } - batch_tags.insert(resolved_tags[i]); - } - } - if (holder == nullptr) { - // NewObjectArray failed (OOM/local-ref exhaustion) - the FollowReferences call below - // (which would have discovered this batch's children) never runs. - ctx.truncated = true; - } else if (!ctx.truncated) { - // A single FollowReferences over the holder array expands this whole BFS level in one - // stop-the-world HeapWalkOperation (instead of one per frontier entry). - ctx._last_visited_batch_tag = 0; // reset rolling cursor - completed_batch_tags.clear(); - u64 follow_start_ticks = TSC::ticks(); - jvmtiError follow_err = - jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); - *safepoint_ticks += TSC::ticks() - follow_start_ticks; - if (follow_err != JVMTI_ERROR_NONE) { - ctx.truncated = true; - } - } - } - } - - if (!ctx.truncated) { - // The whole batch had all its direct children admitted this level: dead entries are pruned, - // live ones are marked EXPANDED, and all are popped off the front. - for (jlong tag : candidate_tags) { - if (live.find(tag) == live.end()) { - _frontier->clear(tag); - } else { - _frontier->markExpanded(tag); - } - source.pop_front(); - } - progress = true; - } else if (!completed_batch_tags.empty()) { - // ROLLING RESUME (order-independent): FollowReferences truncated mid-batch, but the callback - // recorded exactly which batch entries it finished visiting (see - // ReferenceChainPassContext::_completed_batch_tags). - std::vector keep; - keep.reserve(candidate_tags.size()); - for (jlong tag : candidate_tags) { - if (completed_batch_tags.count(tag) != 0) { - if (live.find(tag) == live.end()) { - _frontier->clear(tag); - } else { - _frontier->markExpanded(tag); - } - } else { - keep.push_back(tag); - } - source.pop_front(); - } - // Re-queue the unvisited remainder at the front, preserving input order (push_front in - // reverse). - for (size_t i = keep.size(); i-- > 0;) { - source.push_front(keep[i]); - } - } - // else truncated with no batch entry visited (e.g. GetObjectsWithTags error, holder allocation - // failure, or truncation before the first batch entry was reached): leave the entire batch at - // the front of the source queue for a later pass to retry, same as before. - - if (from_priority) { - // This batch popped entries off _priority_expand's front (or, on truncation, was left - // untouched) - re-derive the membership index from the deque's current contents either way so - // isQueuedForRotation() stays exact for the rotation collectors that run later in this same - // pass. - _priority_expand_set.rebuildFrom(_priority_expand); - } - - if (holder != nullptr) { - jni->DeleteLocalRef(holder); - } - if (jni != nullptr) { - for (jint i = 0; i < resolved_count; i++) { - jni->DeleteLocalRef(resolved_objects[i]); - } - } - if (resolved_objects != nullptr) { - jvmti->Deallocate((unsigned char *)resolved_objects); - } - if (resolved_tags != nullptr) { - jvmti->Deallocate((unsigned char *)resolved_tags); - } - } - - // object_class is NOT deleted here - it is now cached in _cached_object_class and reused across - // calls on this same JNIEnv (see above), not a per-call local ref. - - if (!ctx.truncated && jni != nullptr && object_class == nullptr && - (!_pending_expand.empty() || !_priority_expand.empty())) { - // FindClass("java/lang/Object") failed for this (attached) JNIEnv, so the batching loop above - // never ran even though pending frontier work remains. - ctx.truncated = true; - } - - *edges_admitted = ctx.edges_admitted; - *truncated = ctx.truncated; - *frontier_cap_hit = ctx.frontier_cap_hit; -} - -void ReferenceChainTracker::admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, - int hop_cap, int budget, - int *edges_admitted, - bool *truncated, - bool *frontier_cap_hit, - bool *cycle_complete, - u64 *safepoint_ticks) { - assert(!t_inGCCallback && - "GetLoadedClasses/FollowReferences are JVMTI Heap-category calls " - "and must not be made from GarbageCollectionStart/Finish"); - *edges_admitted = 0; - *truncated = false; - *frontier_cap_hit = false; - *cycle_complete = false; - - if (jni == nullptr) { - // No JNIEnv to build the holder array on (some test seams) - see expandFrontier()'s own - // identical guard. - return; - } - - jint class_count = 0; - jclass *classes = nullptr; - jvmtiError classes_err = jvmti->GetLoadedClasses(&class_count, &classes); - if (classes_err != JVMTI_ERROR_NONE) { - return; - } - if (class_count <= 0) { - if (classes != nullptr) { - jvmti->Deallocate((unsigned char *)classes); - } - return; - } - - // GetLoadedClasses() gives no ordering guarantee across separate calls, so the cursor below is - // only meaningful as an index into THIS call's array - reprioritize it every call rather than - // trying to cache an ordering. - jint app_boundary = 0; - for (jint i = 0; i < class_count; i++) { - jobject loader = nullptr; - jvmtiError loader_err = jvmti->GetClassLoader(classes[i], &loader); - bool is_app_class = (loader_err == JVMTI_ERROR_NONE) && (loader != nullptr); - if (loader != nullptr) { - jni->DeleteLocalRef(loader); - } - if (is_app_class) { - if (i != app_boundary) { - std::swap(classes[i], classes[app_boundary]); - } - app_boundary++; - } - } - - // Stable chunk order: GetLoadedClasses() returns an arbitrary order per call, so an index cursor - // over raw call order can MISS classes entirely within a lap (each chunk would cover a different - // random subset). - { - std::vector tags((size_t)class_count, 0); - for (jint i = 0; i < class_count; i++) { - jvmti->GetTag(classes[i], &tags[i]); - } - std::vector order((size_t)class_count); - for (jint i = 0; i < class_count; i++) { - order[i] = i; - } - std::sort(order.begin(), order.begin() + app_boundary, - [&tags](jint a, jint b) { return tags[a] < tags[b]; }); - std::sort(order.begin() + app_boundary, order.end(), - [&tags](jint a, jint b) { return tags[a] < tags[b]; }); - std::vector sorted((size_t)class_count); - for (jint i = 0; i < class_count; i++) { - sorted[i] = classes[order[i]]; - } - memcpy(classes, sorted.data(), (size_t)class_count * sizeof(jclass)); - } - - if (_static_field_sweep_cursor >= class_count) { - // Loaded-class count shrank since the last chunk (classes unloaded) - restart the lap rather - // than reading out of range. - _static_field_sweep_cursor = 0; - _static_field_sweep_cycle_truncated = false; - } - jint chunk_start = _static_field_sweep_cursor; - jint chunk_end = - std::min(chunk_start + STATIC_FIELD_SWEEP_CHUNK_CLASSES, class_count); - jint chunk_count = chunk_end - chunk_start; - - // Same java/lang/Object element-type cache expandFrontier() uses for its own frontier-holder - // array - shared across both call sites on this same attached JNIEnv rather than a second - // FindClass() per pass. - if (_cached_object_class == nullptr) { - jclass local = jni->FindClass("java/lang/Object"); - if (!jniExceptionCheck(jni) && local != nullptr) { - _cached_object_class = (jclass)jni->NewGlobalRef(local); - } - if (local != nullptr) { - jni->DeleteLocalRef(local); - } - } - jclass object_class = _cached_object_class; - - if (object_class == nullptr || - jni->EnsureLocalCapacity(class_count + 16) < 0 || - jniExceptionCheck(jni)) { - for (jint i = 0; i < class_count; i++) { - jni->DeleteLocalRef(classes[i]); - } - jvmti->Deallocate((unsigned char *)classes); - return; - } - - jobjectArray holder = jni->NewObjectArray(chunk_count, object_class, nullptr); - if (jniExceptionCheck(jni)) { - // OutOfMemoryError (or any other exception) building the holder - clear it rather than let it - // survive into the DeleteLocalRef() calls below (JNI spec: undefined behavior with a pending - // exception across ordinary JNI calls), same as expandFrontier()'s identical case. - holder = nullptr; - } - if (holder != nullptr) { - // Fill in REVERSE chunk order: holder[0] = classes[chunk_end-1], ..., holder[chunk_count-1] = - // classes[chunk_start]. - for (jint i = 0; i < chunk_count; i++) { - jni->SetObjectArrayElement(holder, i, classes[chunk_end - 1 - i]); - if (jniExceptionCheck(jni)) { - holder = nullptr; - break; - } - } - } - - // GetLoadedClasses() returned a local ref for every class regardless of chunk selection - free - // all of them here, not just the chunk. - for (jint i = 0; i < class_count; i++) { - jni->DeleteLocalRef(classes[i]); - } - jvmti->Deallocate((unsigned char *)classes); - - if (holder == nullptr) { - // OOM/local-ref exhaustion/array-store failure - skip this pass's sweep rather than treating it - // like the manual walk's own truncation (see this method's own header comment). - return; - } - - ReferenceChainPassContext ctx; - ctx.tracker = this; - ctx.frontier = _frontier; - ctx.hop_cap = hop_cap; - ctx.budget = budget; - ctx.edges_admitted = 0; - ctx.truncated = false; - ctx.frontier_cap_hit = false; - // Empty (not null) batch_tags forces heapReferenceCallback() to stop at exactly one hop past each - // class - see this method's own header comment for why a deeper descent here would reintroduce - // the whole-graph FollowReferences cost the array-holder batching design otherwise avoids. - std::unordered_set empty_batch_tags; - ctx.batch_tags = &empty_batch_tags; - // Lets heapReferenceCallback() walk past the holder->class seed edge (see - // ReferenceChainPassContext::static_field_seed's own comment) so this sweep actually reaches each - // class's static fields instead of stopping at the negative-tagged class object itself. - ctx.static_field_seed = true; - // Per-class non-STATIC_FIELD admission cap (see ReferenceChainPassContext::_class_other_cap's own - // comment). - ctx._class_other_cap = STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS; - - jvmtiHeapCallbacks callbacks; - memset(&callbacks, 0, sizeof(callbacks)); - callbacks.heap_reference_callback = heapReferenceCallback; - u64 follow_start_ticks = TSC::ticks(); - jvmtiError follow_err = - jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); - *safepoint_ticks += TSC::ticks() - follow_start_ticks; - jni->DeleteLocalRef(holder); - if (follow_err != JVMTI_ERROR_NONE) { - return; - } - - *edges_admitted = ctx.edges_admitted; - *truncated = ctx.truncated; - *frontier_cap_hit = ctx.frontier_cap_hit; - - if (ctx.truncated) { - _static_field_sweep_cycle_truncated = true; - // Resumable cursor: instead of skipping to chunk_end (losing every class after the interruption - // point for the rest of this lap), redo the chunk on the next pass. - _static_field_sweep_cursor = chunk_start; - } else { - // Full advance: every class in the chunk was processed. - _static_field_sweep_cursor = chunk_end; - } - if (_static_field_sweep_cursor >= class_count) { - *cycle_complete = !_static_field_sweep_cycle_truncated; - _static_field_sweep_cursor = 0; - _static_field_sweep_cycle_truncated = false; - } -} - -bool ReferenceChainTracker::releaseSearchTags(jvmtiEnv *jvmti, JNIEnv *jni) { - assert(!t_inGCCallback && - "GetObjectsWithTags is a JVMTI Heap-category call and must not be " - "made from GarbageCollectionStart/Finish"); - if (jvmti == nullptr || _frontier == nullptr) { - return true; // nothing to release - } - - jlong scan_limit = _frontier->size(); - std::vector live_tags; - for (jlong tag = 1; tag <= scan_limit; tag++) { - FrontierEntry entry{}; - if (_frontier->lookup(tag, &entry) && - entry.state != FrontierEntryState::ABANDONED) { - live_tags.push_back(tag); - } - } - if (live_tags.empty()) { - return true; - } - - jint resolved_count = 0; - jobject *resolved_objects = nullptr; - jlong *resolved_tags = nullptr; - if (jvmti->GetObjectsWithTags((jint)live_tags.size(), live_tags.data(), - &resolved_count, &resolved_objects, - &resolved_tags) != JVMTI_ERROR_NONE) { - // GetObjectsWithTags() itself failed (e.g. JVMTI_ERROR_OUT_OF_MEMORY): we do NOT know which, if - // any, of live_tags are still live objects, so do not mark any of them ABANDONED here - doing - // so while their JVMTI tag might still be set would let a restarted search's nextTag() sequence - // eventually reissue the same numeric tag to a brand-new object, corrupting FrontierTable's - // tag-uniqueness invariant (see this method's own header comment). - Counters::increment(REFERENCE_CHAIN_TAG_RELEASE_FAILED); - Log::warn("ReferenceChains: GetObjectsWithTags failed while releasing " - "%zu search tag(s); will retry before allowing a search " - "restart", - live_tags.size()); - return false; - } - - for (jint i = 0; i < resolved_count; i++) { - // clearTag() rather than a raw SetTag() call - reuses the same helper (and its GC-callback - // self-consistency assert) tagObject/ getTag already go through. - clearTag(jvmti, resolved_objects[i]); - if (jni != nullptr) { - jni->DeleteLocalRef(resolved_objects[i]); - } - } - if (resolved_objects != nullptr) { - jvmti->Deallocate((unsigned char *)resolved_objects); - } - if (resolved_tags != nullptr) { - jvmti->Deallocate((unsigned char *)resolved_tags); - } - // Tags that failed to resolve above are already dead (JVMTI forgot them with their object) - - // nothing to release, just mark the record ABANDONED below like every other entry this search - // owned. - for (jlong tag : live_tags) { - _frontier->clear(tag); - } - return true; -} - -bool ReferenceChainTracker::runPass(jvmtiEnv *jvmti, JNIEnv *jni, - bool *out_truncated) { - if (!_enabled || jvmti == nullptr || _frontier == nullptr) { - TEST_LOG_SUMMARY("ReferenceChainTracker::runPass early-exit: enabled=%d jvmti=%p frontier=%p", - _enabled, (void *)jvmti, (void *)_frontier); - return false; - } - - if (_search_state != SearchState::RUNNING) { - // The search already reached a terminal outcome - nothing left for another pass to do until - // shouldRunPass() decides to restartSearch() (this class's header comment), which flips - // _search_started back to false before this method is called again. - if (!_tags_released) { - _tags_released = releaseSearchTags(jvmti, jni); - } - TEST_LOG_SUMMARY("ReferenceChainTracker::runPass no-op: searchState=%d already terminal " - "tagsReleased=%d", - (int)_search_state, _tags_released); - if (out_truncated != nullptr) { - *out_truncated = false; - } - return true; - } - - resolveLoadedClasses(jvmti, jni); - - TEST_LOG_SUMMARY("ReferenceChainTracker::runPass starting JVMTI walk: " - "search_started=%d frontierSize=%zu", - _search_started, _frontier != nullptr ? _frontier->size() : (size_t)0); - - int edges_admitted = 0; - bool truncated = false; - bool frontier_cap_hit = false; - jvmtiError err; - // Whole-call wall-clock duration of runPassManualWalk() below - includes root/stack-ref - // enumeration dispatch, frontier-table bookkeeping, and rotation-candidate collection, in - // addition to the actual in-safepoint JVMTI calls. - u64 pass_wall_ticks = 0; - // Genuine in-safepoint cost of this pass, accumulated by runPassManualWalk() across every - // IterateOverReachableObjects/ FollowReferences call it makes (root enum, static-field sweep, - // ordinary expansion, rotation re-expansion) - explicitly excluding GetObjectsWithTags (not a - // safepoint call) and every bookkeeping line in between. - u64 safepoint_ticks = 0; - - // Every pass is driven by the manual walk (runPassManualWalk() - IterateOverReachableObjects for - // roots, then a batched array-holder FollowReferences per BFS level in expandFrontier()), on - // every collector. - bool manual_first_pass = !_search_started; - if (manual_first_pass) { - _search_started = true; - store(_search_start_ns, OS::nanotime()); - } - - // Root/stack-ref enumeration alone never discovers a root's transitive children - // (runPassManualWalk()'s own comment) - there is no "first pass walks the whole graph inline" - // shortcut here, so every pass (first or resumed) takes the same expand-frontier shape. - u64 now_ns = OS::nanotime(); - bool run_root_enum = manual_first_pass || _root_enum_truncated_last_time || - (now_ns - _last_root_enum_ns >= ROOT_ENUM_MIN_INTERVAL_NS); - - int frontier_size_before_pass = _frontier != nullptr ? _frontier->size() : 0; - - u64 call_start_ticks = TSC::ticks(); - runPassManualWalk(jvmti, jni, run_root_enum, _first_pass_budget, - _effective_budget, &edges_admitted, &truncated, - &frontier_cap_hit, &safepoint_ticks); - pass_wall_ticks = TSC::ticks() - call_start_ticks; - // TSC::ticks() is monotonic but not necessarily free of measurement noise between the outer - // call_start_ticks snapshot and the several inner TSC::ticks() snapshots safepoint_ticks is built - // from - clamp rather than underflow if the accumulated safepoint portion ever reads back larger - // than the whole-call wall time it's a subset of. - u64 non_safepoint_ticks = - pass_wall_ticks > safepoint_ticks ? pass_wall_ticks - safepoint_ticks : 0; - err = JVMTI_ERROR_NONE; - - store(_passes_run, load(_passes_run) + 1); - _last_pass_gc_finish_epoch = gcFinishEpoch(); - store(_last_pass_ns, OS::nanotime()); - if (!run_root_enum) { - // A pass that ran root/stack-ref enumeration spends _first_pass_budget, not _effective_budget - - // its duration is not a signal about the per-pass cost updatePacing() is trying to regulate - // (expandFrontier()'s cheap, per-node expansion calls), so feeding it in here would throttle - // _effective_budget down for every one of those unrelated later passes based on a single, - // deliberately oversized outlier. - updatePacing(safepoint_ticks); - } else { - // Excluded from the budget/cadence controller above, but not from the borrow ceiling's - // revocation check (see maybeRevokeBorrowForRootEnumPass()'s own comment) - a root-enum pass's - // in-safepoint cost is real pause time and must still be able to revoke a borrowed-budget grant - // the pacing controller would otherwise keep believing is safe. - maybeRevokeBorrowForRootEnumPass(safepoint_ticks); - } - // accumulate this pass's own in-safepoint cost toward the running total restartSearch() will - // spend into _safepoint_pain_budget once the search reaches a terminal state - same - // TSC::ticks_to_millis() conversion updatePacing() already uses for its own pass-duration signal. - _search_pain_ms += TSC::ticks_to_millis(safepoint_ticks); - // Independent leaky bucket for the non-safepoint remainder of this pass (root/stack-ref - // enumeration dispatch, frontier-table admission, rotation-candidate collection) - see - // _cpu_pain_budget's own comment (referenceChains.h) for why this needs to be tracked separately - // from both _safepoint_pain_budget above and _pause_pid's safepoint_ticks signal. - _cpu_pain_budget.spend(TSC::ticks_to_millis(non_safepoint_ticks)); - - // Apply terminal conditions in priority order. - bool has_pending_frontier = truncated; - int frontier_size_after = _frontier->size(); - if (frontier_cap_hit) { - // Frontier table is full -- no new entries can ever be admitted, so frontier_size_after can - // never exceed frontier_size_before_pass again. - store(_abandon_reason, (u8)SearchAbandonReason::FRONTIER_CAP); - storeRelease(_search_state, (u8)SearchState::ABANDONED); - enqueuePendingAbandonedEvent(); - TEST_LOG_SUMMARY("ReferenceChainTracker::runPass frontier cap hit -- " - "abandoning search (size=%d)", - frontier_size_after); - } else if (!has_pending_frontier && _watched_leak_klass_count == 0) { - storeRelease(_search_state, (u8)SearchState::COMPLETED); - } else if (_ttl_ms > 0 && - TSC::ticks_to_millis(OS::nanotime() - load(_search_start_ns)) >= - (u64)_ttl_ms) { - // TTL bounds stop-the-world work independently of frontier progress. - store(_abandon_reason, (u8)SearchAbandonReason::TTL); - storeRelease(_search_state, (u8)SearchState::ABANDONED); - enqueuePendingAbandonedEvent(); - TEST_LOG_SUMMARY("ReferenceChainTracker::runPass ttl expired -- abandoning search " - "(ttl=%ldms elapsed_ms=%llu)", - _ttl_ms, - (unsigned long long)TSC::ticks_to_millis(OS::nanotime() - - load(_search_start_ns))); - } else if (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && - !isUrgent()) { - // The frontier hasn't grown for NO_PROGRESS_PASS_LIMIT consecutive passes — the search is - // genuinely stuck (not just slow), so abandon. - store(_abandon_reason, (u8)SearchAbandonReason::TTL); - storeRelease(_search_state, (u8)SearchState::ABANDONED); - enqueuePendingAbandonedEvent(); - } else if (_candidate_count > 0 && - __builtin_popcountll(_candidate_found_bits) == - (u64)_candidate_count) { - // Canary early termination: all leaked candidates have been found -- the search is complete. - storeRelease(_search_state, (u8)SearchState::COMPLETED); - Counters::increment(REFERENCE_CHAIN_CANDIDATES_FOUND, - __builtin_popcountll(_candidate_found_bits)); - } else if (_candidate_count > 0 && - _passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && - _passes_since_last_candidate_progress >= - canaryStuckPassLimit()) { - // Canary-specific stuck detector - deliberately NOT suppressed by isUrgent() (contrast the - // ordinary TTL check above). - store(_abandon_reason, (u8)SearchAbandonReason::CANARY_STUCK); - storeRelease(_search_state, (u8)SearchState::ABANDONED); - enqueuePendingAbandonedEvent(); - if (_canary_stuck_restart_count < MAX_CANARY_STUCK_BACKOFF_SHIFT) { - _canary_stuck_restart_count++; - } - } - - // Track progress: if the frontier grew this pass, reset the no-progress counter. - if (frontier_size_after > frontier_size_before_pass) { - _passes_since_last_progress = 0; - } else { - _passes_since_last_progress++; - } - - // Track canary-specific progress separately - see _passes_since_last_candidate_progress's own - // comment for why frontier growth above does not substitute for this. - { - u64 pass_wall_ms = (u64)TSC::ticks_to_millis(pass_wall_ticks); - _canary_pass_ema_ms = _canary_pass_ema_ms == 0 - ? pass_wall_ms - : _canary_pass_ema_ms * 4 / 5 + pass_wall_ms / 5; - } - int candidate_progress_mark = - _candidate_count + (int)__builtin_popcountll(_candidate_found_bits); - if (candidate_progress_mark > _last_candidate_progress_mark) { - _last_candidate_progress_mark = candidate_progress_mark; - _passes_since_last_candidate_progress = 0; - // Real chase progress (a candidate found or a new one admitted) - the canary lane gets its - // back-to-back spacing back (multiplier 1, see _canary_backoff_mult's own comment). - _canary_backoff_mult = 1; - } else { - _passes_since_last_candidate_progress++; - // No chase progress: double the canary lane's work-scaled spacing multiplier, capped. - if (_candidate_count > 0 && - __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count) { - _canary_backoff_mult = - std::min(_canary_backoff_mult * 2, CANARY_BACKOFF_MULT_MAX); - _last_canary_pass_ns = OS::nanotime(); - } - } - - if (load(_search_state) != SearchState::RUNNING) { - _tags_released = releaseSearchTags(jvmti, jni); - if (_candidate_count > 0) { - // No marker-tag release pass: the marker->leak-tag migration retired - // pre-tagged candidate representatives (nothing sets _candidate_tags - // anymore), so there are no per-candidate marker JVMTI tags to clear - - // releaseSearchTags() above owns every live tag this search minted. - // (The old GetObjectsWithTags(1, &_candidate_tags[i]) loop here was - // worse than dead: with every _candidate_tags[i] left at 0 it asked - // JVMTI to enumerate ALL UNTAGGED objects - potentially the whole - // heap - once per candidate slot on every search stop.) - _candidate_count = 0; - _candidate_found_bits = 0; - memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); - memset(_candidate_qualifying_tid_count, 0, - sizeof(_candidate_qualifying_tid_count)); - _passes_since_last_candidate_progress = 0; - } - // Only CANARY_STUCK should keep escalating canaryStuckPassLimit() - any other terminal reason - // (natural completion, all candidates found, frontier cap, TTL) is an unrelated outcome for - // this chase sequence, so a fresh restart afterward should start back at the base limit. - if (load(_abandon_reason) != SearchAbandonReason::CANARY_STUCK) { - _canary_stuck_restart_count = 0; - } - } - - if (out_truncated != nullptr) { - *out_truncated = truncated; - } - - TEST_LOG_SUMMARY("ReferenceChainTracker::runPass done: err=%d edges_admitted=%d truncated=%d " - "frontier_cap_hit=%d searchState=%d abandonReason=%d frontierSize=%d " - "effectiveBudget=%d effectiveCadenceNs=%llu pendingExpand=%zu priorityExpand=%zu " - "candidateFound=%d/%d discoveredCounts=[%d,%d,%d,%d,%d]", - (int)err, edges_admitted, truncated, frontier_cap_hit, (int)load(_search_state), - (int)_abandon_reason, _frontier->size(), _effective_budget, - (unsigned long long)_effective_cadence_ns, - _pending_expand.size(), _priority_expand.size(), - (int)__builtin_popcountll(_candidate_found_bits), _candidate_count, - _candidate_count > 0 ? _candidate_discovered_count[0] : 0, - _candidate_count > 1 ? _candidate_discovered_count[1] : 0, - _candidate_count > 2 ? _candidate_discovered_count[2] : 0, - _candidate_count > 3 ? _candidate_discovered_count[3] : 0, - _candidate_count > 4 ? _candidate_discovered_count[4] : 0); - - return err == JVMTI_ERROR_NONE; -} - -// Pause-time-SLO feedback loop (see this method's declaration in - -void ReferenceChainTracker::updatePacing(u64 pass_wall_ticks) { - // Truncating to whole milliseconds matches every other PidController usage in this codebase - // (ObjectSampler/MallocTracer/NativeSocketSampler all feed it integer counts, pidController.h's - // `compute(u64 input, ...)`) - sub-ms precision is not meaningful against a millisecond-scale - // target anyway. - u64 pass_ms = TSC::ticks_to_millis(pass_wall_ticks); - // time_delta_coefficient is deliberately 1.0, not a real-elapsed-time ratio - unlike - // ObjectSampler's usage (objectSampler.cpp), which rescales an event count accumulated over a - // variable-length real-time window against a fixed-real-time target, _pause_pid was constructed - // with sampling_window=1 (its own constructor comment above, in start()): one compute() call *is* - // one pass, and pass_ms already IS the per-call quantity being compared against the per-call - // ceiling _target encodes. - double signal = _pause_pid.compute(pass_ms, 1.0); - - // Budget-borrowing (referenceChains.h's _borrowed_budget comment): only a sustained run of - // comfortably-under-target passes earns extra headroom above _budget, and any pass that is not - // comfortably under target revokes it immediately - _budget itself must stay the ceiling the - // instant this search stops proving it has pause-time room to spare. - bool comfortably_under_target = - _effective_pause_target_ms > 0 && - (double)pass_ms <= (double)_effective_pause_target_ms * BORROW_UNDER_TARGET_FRACTION; - if (comfortably_under_target) { - if (_consecutive_under_target_passes < BORROW_WARMUP_PASSES) { - _consecutive_under_target_passes++; - } - if (_consecutive_under_target_passes >= BORROW_WARMUP_PASSES) { - int64_t max_borrow = (int64_t)_budget * (BORROW_CEILING_MULTIPLIER - 1); - int64_t grown = _borrowed_budget + - (int64_t)std::llround((double)_budget * BORROW_GROWTH_FRACTION); - _borrowed_budget = std::min(grown, max_borrow); - } - } else { - _consecutive_under_target_passes = 0; - _borrowed_budget = 0; - } - - int64_t ceiling = (int64_t)_budget + _borrowed_budget; - int64_t floor = ceiling > 0 ? std::min((int64_t)MIN_EFFECTIVE_BUDGET, ceiling) - : 0; - int64_t desired = (int64_t)_effective_budget + (int64_t)std::lround(signal); - int64_t clamped = std::max(floor, std::min(ceiling, desired)); - // Whatever part of `desired` the clamp above could not absorb - positive when there was more - // headroom than the ceiling allows, negative when the pass is still over the pause-time target - // even at the floor. - int64_t overflow = desired - clamped; - _effective_budget = (int)clamped; - - if (overflow < 0) { - // Still over the pause-time ceiling even at the minimum budget - widen the fallback interval - // instead of shrinking the budget further. - u64 step = (u64)(-overflow) * CADENCE_NS_PER_EDGE_OVERFLOW; - _effective_cadence_ns = - std::min(_effective_cadence_ns + step, MAX_EFFECTIVE_CADENCE_NS); - } else if (overflow > 0) { - // Comfortably under the ceiling even at the maximum (config) budget - relax the fallback - // interval. - u64 step = (u64)overflow * CADENCE_NS_PER_EDGE_OVERFLOW; - _effective_cadence_ns = - step >= _effective_cadence_ns - ? MIN_EFFECTIVE_CADENCE_NS - : std::max(_effective_cadence_ns - step, MIN_EFFECTIVE_CADENCE_NS); - } - // overflow == 0: the budget clamp alone fully absorbed this pass's correction - leave the cadence - // at its current value. -} - -// A root/stack-ref enumeration pass never reaches updatePacing() above (see runPass()'s own comment -// on why its wall-clock cost is excluded from the per-pass PID/effective-budget signal), but it -// still spends real pause-time-SLO time. -void ReferenceChainTracker::maybeRevokeBorrowForRootEnumPass( - u64 pass_wall_ticks) { - if (_effective_pause_target_ms <= 0) { - return; - } - u64 pass_ms = TSC::ticks_to_millis(pass_wall_ticks); - bool comfortably_under_target = - (double)pass_ms <= (double)_effective_pause_target_ms * BORROW_UNDER_TARGET_FRACTION; - if (!comfortably_under_target) { - _consecutive_under_target_passes = 0; - _borrowed_budget = 0; - // The ceiling updatePacing() would compute right now collapses to _budget alone (no - // _borrowed_budget term above) - re-clamp _effective_budget immediately instead of leaving the - // borrow-inflated value in place until the next ordinary pass's updatePacing() call. - _effective_budget = std::min(_effective_budget, (int)_budget); - } -} - diff --git a/ddprof-lib/src/main/cpp/referenceChainWalk.cpp b/ddprof-lib/src/main/cpp/referenceChainWalk.cpp deleted file mode 100644 index a74ecfbc38..0000000000 --- a/ddprof-lib/src/main/cpp/referenceChainWalk.cpp +++ /dev/null @@ -1,888 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -#include "referenceChains.h" -#include "referenceChainInternal.h" -#include "common.h" -#include "counters.h" -#include "jniHelper.h" -#include "jvmThread.h" -#include "livenessTracker.h" -#include "log.h" -#include "objectSampler.h" -#include "os.h" -#include "profiler.h" -#include "rcDebugLevel.h" -#include "tsc.h" -#include "vmEntry.h" -#include -#include -#include -#include -#include -#include -#include -#include - -// Heap-walk engine - -void ReferenceChainTracker::resolveLoadedClasses(jvmtiEnv *jvmti, - JNIEnv *jni) { - // Profiler::start() resets the class-name StringDictionary (_class_map.clearAll(), profiler.cpp) - // whenever `reset || _start_time == 0` - which restarts its id namespace at 1, but does NOT touch - // any class's JVMTI-level class-object tag (JVM-level state, unrelated to our dictionary). - u64 current_generation = Profiler::instance()->classMap()->generation(); - bool class_map_reset = current_generation != _last_class_map_generation; - if (class_map_reset) { - TEST_LOG_SUMMARY("ReferenceChainTracker::resolveLoadedClasses class_map generation " - "changed: old=%llu new=%llu - clearing _class_tags and " - "candidate klass_ids may be stale", - (unsigned long long)_last_class_map_generation, - (unsigned long long)current_generation); - _class_tags.clear(); - // Force the scan below to run even if GetLoadedClasses()'s count happens to match the last-seen - // count - -1 can never equal `class_count` (always >= 0), unlike 0 which is a legitimate "no - // classes loaded yet" starting value. - _last_resolved_class_count = -1; - _last_class_map_generation = current_generation; - } - - jclass *classes = nullptr; - jint class_count = 0; - if (jvmti->GetLoadedClasses(&class_count, &classes) != JVMTI_ERROR_NONE || - classes == nullptr) { - return; - } - - // Skip the per-class GetTag()/GetClassSignature() scan entirely once the loaded-class count has - // not CHANGED since the last time this ran it: every already-tagged class stays tagged forever - // (tags are never cleared once assigned - see _class_tags' own comment), so a resumed pass with - // no newly-loaded classes has nothing left to resolve. - if (class_count != _last_resolved_class_count) { - for (jint i = 0; i < class_count; i++) { - jclass klass = classes[i]; - jlong tag = 0; - // Resolve if not yet tagged (ordinary case: a newly-loaded class), or unconditionally on a - // class-map reset (class_map_reset above) - a class already tagged from a prior generation - // still carries that same JVMTI tag (untouched by clearAll()), but the dictionary id it used - // to map to is gone, so its name must be re-resolved into the new generation too. - if (jvmti->GetTag(klass, &tag) == JVMTI_ERROR_NONE && - (tag == 0 || class_map_reset)) { - // Resolve its name now, via the same GetClassSignature + normalizeClassSignature + - // Profiler::lookupClass sequence ObjectSampler::recordAllocation() already uses - // (objectSampler.cpp:76-90), reused rather than re-derived. - char *class_name = nullptr; - if (jvmti->GetClassSignature(klass, &class_name, nullptr) == - JVMTI_ERROR_NONE && - class_name != nullptr) { - const char *name_slice = nullptr; - size_t name_len = 0; - if (ObjectSampler::normalizeClassSignature(class_name, &name_slice, - &name_len)) { - int id = Profiler::instance()->lookupClass(name_slice, name_len); - if (id != -1) { - TEST_LOG("ReferenceChainTracker::resolveClassMap id=%d name=%.*s", - id, (int)name_len, name_slice); - // Reuse the existing tag if this class was already tagged by a prior generation - - // only the resolved id needs refreshing, not the tag identity heapReferenceCallback() - // keys off of. - jlong class_tag = tag != 0 ? tag : nextClassTag(); - if (tag != 0 || - jvmti->SetTag(klass, class_tag) == JVMTI_ERROR_NONE) { - if (tag == 0) { - // Adopt the tag actually installed on the class object: LivenessTracker's - // mintStableClassTagIfNeeded() may have installed its own tag between our GetTag - // and SetTag (two SetTag calls on the same untagged class - the last writer wins - // on the class). - jlong installed = 0; - if (jvmti->GetTag(klass, &installed) == JVMTI_ERROR_NONE && - installed != 0) { - class_tag = installed; - } - } - _class_tags.insert(class_tag, (u32)id); - } - } - } - jvmti->Deallocate((unsigned char *)class_name); - } - } - // GetLoadedClasses() hands back class_count fresh JNI local refs - delete each immediately - // rather than holding all of them alive at once, since class_count can run into the - // thousands. - if (jni != nullptr) { - jni->DeleteLocalRef(klass); - } - } - _last_resolved_class_count = class_count; - } else if (jni != nullptr) { - // Still owe DeleteLocalRef for every fresh local ref GetLoadedClasses() just handed back, even - // though the scan above was skipped. - for (jint i = 0; i < class_count; i++) { - jni->DeleteLocalRef(classes[i]); - } - } - jvmti->Deallocate((unsigned char *)classes); -} - -jint JNICALL ReferenceChainTracker::heapReferenceCallback( - jvmtiHeapReferenceKind reference_kind, - const jvmtiHeapReferenceInfo *reference_info, jlong class_tag, - jlong referrer_class_tag, jlong size, jlong *tag_ptr, - jlong *referrer_tag_ptr, jint length, void *user_data) { - ReferenceChainPassContext *ctx = (ReferenceChainPassContext *)user_data; - - if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { - // stopThread() has set this right before pthread_kill()/pthread_join() - see that method's own - // comment. - ctx->truncated = true; - return JVMTI_VISIT_ABORT; - } - - if (ctx->tracker->_pass_deadline_ns != 0 && - (++ctx->deadline_check_counter & 0xFFF) == 0 && - OS::nanotime() >= ctx->tracker->_pass_deadline_ns) { - // This pass has run past its wall-clock share (see _pass_deadline_ns's own comment) - treat it - // exactly like ordinary budget exhaustion so it ends early without abandoning the search; a - // later pass re-enumerates whatever roots/edges this one didn't get to. - ctx->truncated = true; - return JVMTI_VISIT_ABORT; - } - - // Retention-edge identity for every admission site below: the JVMTI heap callback's field ordinal - // (the JVMTI-SPECIFICATION numbering over the referrer's flattened field space - see - // FrontierEntry:: referrer_field_index's own comment) for FIELD/STATIC_FIELD edges, -1 otherwise; - // and the referrer's class tag when the referrer is a CLASS OBJECT (root-attached static edges - - // interior hops get their referrer class from the parent entry at chain-reconstruction time, so - // only the parent_tag==0 case needs it recorded here). - jint edge_field_index = -1; - if (reference_info != nullptr && - (reference_kind == JVMTI_HEAP_REFERENCE_FIELD || - reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD)) { - edge_field_index = reference_info->field.index; - } - jlong edge_referrer_class_tag = 0; - if (referrer_tag_ptr != nullptr && *referrer_tag_ptr < 0) { - edge_referrer_class_tag = *referrer_tag_ptr; - } - - // NOTE: the retired canary marker-tag decode used to live here (objects - // pre-tagged with MARKER_TAG_BASE - i were pruned as leaves and recorded as - // chain roots). The marker->leak-tag migration stopped pre-tagging candidate - // representatives entirely - no JVMTI tag in the process can ever be - // <= MARKER_TAG_BASE (-2^62): leak tags are positive (LEAK_TAG_BASE), - // frontier tags positive, class tags small negative magnitudes - so the - // branch was unreachable. Candidate discovery now runs exclusively through - // the leak-tag interception in pollWatchedTargets() (which also records - // _candidate_found_bits/_candidate_frontier_tags). - - if (*tag_ptr < 0) { - if (ctx->static_field_seed && - reference_kind == JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT && - referrer_tag_ptr != nullptr && *referrer_tag_ptr == 0) { - // admitStaticFieldRoots()'s own holder[i] -> class edge: referrer_tag_ptr points at the - // transient, never-tagged seed array itself (tag 0), not at a frontier-admitted parent. - return JVMTI_VISIT_OBJECTS; - } - // Referee is a class object already tagged negative by resolveLoadedClasses() (that pre-pass - // runs before FollowReferences in runPass(), so every loaded class already carries a negative - // tag by this point). - return 0; - } - if (reference_kind == JVMTI_HEAP_REFERENCE_CLASS || - reference_kind == JVMTI_HEAP_REFERENCE_SYSTEM_CLASS) { - // Definitionally a class by reference_kind (CLASS: "reference from an object to its class"; - // SYSTEM_CLASS: a root reference to a class) even if resolveLoadedClasses() failed to - // resolve/tag this particular one (e.g. a transient StringDictionary contention failure) and - // its tag is therefore not yet negative. - return 0; - } - - if (ctx->truncated) { - // Defensive: FollowReferences should already have stopped delivering callbacks after a - // JVMTI_VISIT_ABORT return below; this just avoids doing further work if one more callback - // arrives anyway. - return JVMTI_VISIT_ABORT; - } - - jlong parent_tag = 0; - u32 depth = 0; - if (referrer_tag_ptr != nullptr) { - jlong rtag = *referrer_tag_ptr; - if (rtag > 0) { - FrontierEntry parent{}; - if (ctx->frontier->lookup(rtag, &parent)) { - parent_tag = rtag; - depth = parent.depth + 1; - } - // lookup() failing for a positive rtag should not happen - a referrer must already be one of - // our tagged frontier objects for its own outgoing edges to be traversed at all - // (FollowReferences only explores past an object this callback returned JVMTI_VISIT_OBJECTS - // for) - but fall back to root-like (parent_tag=0/depth=0) rather than corrupt the chain if - // it ever does. - } - // rtag < 0: referrer is a pre-tagged class object (e.g. a static field holding this reference) - // - treated as root-like rather than attributed to a parent hop, since class objects are never - // admitted as frontier entries and so have no depth/parent_tag of their own (see the *tag_ptr < - // 0 check above). - } - // referrer_tag_ptr == nullptr: a heap-root reference (JNI global, thread stack local/JNI local, - // monitor, thread, system class, ...) - parent_tag and depth stay 0. - - if (depth >= (u32)ctx->hop_cap) { - // Hop cap: do not admit this object into the frontier, and do not expand further from it - - // enforced here rather than discovering-then-discarding. - return 0; - } - - if (ctx->static_field_seed && referrer_tag_ptr != nullptr && - *referrer_tag_ptr < 0) { - // Referrer is the class object opened by the static_field_seed branch above. - if (*referrer_tag_ptr != ctx->_seed_class_tag) { - ctx->_seed_class_tag = *referrer_tag_ptr; - ctx->_class_other_admitted = 0; - ctx->_classes_in_chunk_visited++; - } - if (reference_kind != JVMTI_HEAP_REFERENCE_STATIC_FIELD) { - if (ctx->_class_other_cap > 0 && - ctx->_class_other_admitted >= ctx->_class_other_cap) { - // Quota exhausted for this class - drop the edge. Count every drop, and count the first - // drop for this class separately so the two counters together distinguish "a few fat - // outlier classes dropping many edges" from "systematic drops across almost all classes" - // (cap too low). - Counters::increment(REFERENCE_CHAIN_STATIC_SWEEP_NON_STATIC_DROPPED); - if (ctx->_class_other_admitted == ctx->_class_other_cap) { - Counters::increment(REFERENCE_CHAIN_STATIC_SWEEP_CLASSES_CAPPED); - } - return 0; - } - ctx->_class_other_admitted++; - } - } - - // Leak tag: this object was directly tagged by LivenessTracker's tagLeakInstances() because it's - // a tracked leaking object. - if (isLeakTag(*tag_ptr)) { - jlong leak_tag = *tag_ptr; - // Allocate a frontier tag for this object - jlong frontier_tag = ctx->tracker->nextTag(); - u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); - u8 root_kind = parent_tag == 0 ? (u8)reference_kind : 0; - if (ctx->frontier->insert(frontier_tag, parent_tag, referrer_klass, - depth, FrontierEntryState::FRONTIER, - root_kind, class_tag, edge_field_index, - (u8)reference_kind, - parent_tag == 0 ? edge_referrer_class_tag : 0)) { - // Store the leak tag in the frontier entry - ctx->frontier->setLeakTag(frontier_tag, leak_tag); - *tag_ptr = frontier_tag; - ctx->edges_admitted++; - TEST_LOG("ReferenceChainTracker::heapReferenceCallback leak-tag " - "intercepted: leak_tag=%lld -> frontier_tag=%lld depth=%u " - "parent_tag=%lld", - (long long)leak_tag, (long long)frontier_tag, depth, - (long long)parent_tag); - ctx->tracker->trackLeakAccumulation(ctx->frontier, class_tag, - parent_tag, frontier_tag); - // Index maintenance: a leak-tagged object admitted root-attached by a durable root edge (e.g. - // a static field directly holding a tagged chunk) is the highest-priority anchor tier - // (leak_tag != 0). - if (parent_tag == 0 && - (root_kind == (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || - root_kind == (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL)) { - ctx->tracker->addToStaticAnchorIndex(frontier_tag, class_tag, - root_kind); - } - // Auto-mark: record this as a discovered instance, with eviction rights over uncorrelated - // noise slots (see recordDiscoveredInstance). - if (ctx->tracker->_candidate_count > 0) { - u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); - ctx->tracker->recordDiscoveredInstance(klass_id, frontier_tag, true); - } - } else { - // Frontier cap hit - ctx->truncated = true; - ctx->frontier_cap_hit = true; - return JVMTI_VISIT_ABORT; - } - return JVMTI_VISIT_OBJECTS; - } - - // DESCEND-WALK GATES (no-ops on every ordinary walk - the ReferenceChainPassContext fields below - // are zero-initialized and only descendFromAnchor() sets them). - if (ctx->_no_descend_class_tag_count > 0) { - for (int i = 0; i < ctx->_no_descend_class_tag_count; i++) { - if (ctx->_no_descend_class_tags[i] == class_tag) { - return 0; - } - } - } - if (ctx->_descent_anchor_tag != 0 && ctx->_anchor_descend_class_tag != 0 && - referrer_tag_ptr != nullptr && - *referrer_tag_ptr == ctx->_descent_anchor_tag && - class_tag != ctx->_anchor_descend_class_tag) { - // This descend walk's ANCHOR object's own edge, and the referee is not the gate class (see - // ReferenceChainPassContext::_anchor_descend_class_tag's own comment - e.g. - // walkCandidateThreadLocals() walks ONLY the Thread's ThreadLocalMap edges, never enumerating - // the Thread's other fields). - return 0; - } - - if (*tag_ptr == 0) { - // First time this object is visited in this pass. - u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); - // reference_kind describes this admitting edge; only meaningful for a root-attached entry - // (parent_tag == 0) - see FrontierEntry::root_kind's own comment for why a non-root entry's - // edge kind is not recorded. - u8 root_kind = parent_tag == 0 ? (u8)reference_kind : 0; - ReferenceChainTracker::AdmitResult result = ctx->tracker->admitObject( - ctx->frontier, ctx->hop_cap, ctx->budget, &ctx->edges_admitted, - tag_ptr, parent_tag, referrer_klass, depth, root_kind, class_tag, - ctx->admit_priority, edge_field_index, (u8)reference_kind, - parent_tag == 0 ? edge_referrer_class_tag : 0); - switch (result) { - case ReferenceChainTracker::AdmitResult::BUDGET_EXHAUSTED: - ctx->truncated = true; - return JVMTI_VISIT_ABORT; - case ReferenceChainTracker::AdmitResult::FRONTIER_CAP_HIT: - // Stop this pass when the frontier table is full. - ctx->truncated = true; - ctx->frontier_cap_hit = true; - return JVMTI_VISIT_ABORT; - default: - // ADMITTED, or HOP_CAP/ALREADY_ADMITTED (neither reachable here: the hop-cap check above - // already returned before this branch, and *tag_ptr == 0 rules out ALREADY_ADMITTED) - - // nothing further to do. - break; - } - // Index maintenance: track root-attached durable anchors for O(anchors) collector iteration - // instead of O(frontier_size) table scan. - if (result == ReferenceChainTracker::AdmitResult::ADMITTED && - parent_tag == 0) { - ctx->tracker->addToStaticAnchorIndex(*tag_ptr, class_tag, root_kind); - } - // Auto-mark: if this object's class matches a watched leak class, record its frontier tag so - // pollWatchedTargets() can build a chain event for it. - if (result == ReferenceChainTracker::AdmitResult::ADMITTED && - ctx->tracker->_candidate_count > 0) { - u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); - if (klass_id == 0) { - // class_tag not in _class_tags - either class map rotated (resolveLoadedClasses hasn't - // re-resolved yet) or this class was never tagged. - TEST_LOG("ReferenceChainTracker::auto-mark class_tag=%lld " - "unresolved (not in _class_tags)", - (long long)class_tag); - } else { - bool matched = false; - for (int s = 0; s < ctx->tracker->_candidate_count; s++) { - if (ctx->tracker->_candidate_klass_ids[s] == klass_id) { - matched = true; - ctx->tracker->recordDiscoveredInstance(klass_id, *tag_ptr, - false); - break; - } - } - if (!matched && klass_id != 0) { - // klass_id resolved but doesn't match any candidate - likely class map rotation made - // candidate klass_ids stale - TEST_LOG("ReferenceChainTracker::auto-mark klass_id=%u " - "resolved but no candidate match (candidates=[%u,%u,%u,%u,%u])", - klass_id, - ctx->tracker->_candidate_count > 0 ? ctx->tracker->_candidate_klass_ids[0] : 0, - ctx->tracker->_candidate_count > 1 ? ctx->tracker->_candidate_klass_ids[1] : 0, - ctx->tracker->_candidate_count > 2 ? ctx->tracker->_candidate_klass_ids[2] : 0, - ctx->tracker->_candidate_count > 3 ? ctx->tracker->_candidate_klass_ids[3] : 0, - ctx->tracker->_candidate_count > 4 ? ctx->tracker->_candidate_klass_ids[4] : 0); - } - } - } - } else if (*tag_ptr > 0) { - // Already-tagged object reached via a new edge. This arm - NOT the first-admission block above - // - is where an already-admitted entry's shape can be corrected: - // improveChain/reparentToDurableRoot for a deeper/equal-durable path, - // maybeUpgradeRootAttachedRootKind for a new root-like edge. - if (parent_tag != 0) { - // This new path is deeper - replace the shallow root-attached entry with the deeper - // chain-attached entry. - u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); - // Pre-read the CURRENT shape: if improveChain() below succeeds, this root-attached durable - // entry is about to be replaced with a deeper chain-attached one - i.e. it is leaving the - // population collectStaticFieldAnchorsForRotation() can select, at exactly this moment. - FrontierEntry pre_improve_entry{}; - bool was_root_attached_durable = - ctx->frontier->lookup(*tag_ptr, &pre_improve_entry) && - pre_improve_entry.parent_tag == 0 && - rootKindDurability(pre_improve_entry.root_kind) >= 2; - if (ctx->frontier->improveChain(*tag_ptr, parent_tag, referrer_klass, - depth, 0, edge_field_index, - (u8)reference_kind)) { - // Chain was improved — invalidate any cached chain for this tag so pollWatchedTargets - // rebuilds it with the deeper path. - ctx->tracker->invalidateResolvedChain(*tag_ptr); - if (was_root_attached_durable) { - // Demotion push (B'): the replaced entry's static/JNI-global attribution was its only - // anchor-tier eligibility, and it is gone now. - ctx->tracker->pushAtRiskStaticAnchor( - *tag_ptr, ctx->tracker->classTags()->resolve(class_tag)); - } - } else if (ctx->frontier->reparentToDurableRoot( - *tag_ptr, parent_tag, referrer_klass, edge_field_index, - (u8)reference_kind)) { - // Equal-depth re-parent from a transient root to a durable one (improveChain() cannot - // express it - see its declaration) - same cache invalidation so the rebuilt chain uses the - // durable root. - ctx->tracker->invalidateResolvedChain(*tag_ptr); - } - } else { - // Already-admitted entry reached via a NEW root-like edge (parent_tag == 0): the static-field - // sweep's class -> field edge reports the class as the referrer with a negative tag, which - // the rtag < 0 branch above treats as root-like (class objects are never frontier entries), - // and heap-root references arrive here with referrer_tag_ptr == nullptr. - if (ctx->tracker->maybeUpgradeRootAttachedRootKind(ctx->frontier, - *tag_ptr, - (u8)reference_kind)) { - ctx->tracker->invalidateResolvedChain(*tag_ptr); - } else if (reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) { - // The upgrade refused (maybeUpgradeRootAttachedRootKind returns false for parent_tag != 0 - // by design), so this STATIC_FIELD edge just proved an at-risk static attachment the anchor - // tier's parent_tag == 0 filter can never see: a holder already admitted as a non-root - // child (find-anchor-holder-eviction). - FrontierEntry entry{}; - if (ctx->frontier->lookup(*tag_ptr, &entry) && - entry.parent_tag != 0) { - ctx->tracker->pushAtRiskStaticAnchor( - *tag_ptr, ctx->tracker->classTags()->resolve(class_tag)); - } - } - } - } - - if (ctx->batch_tags != nullptr) { - // ARRAY-HOLDER BATCHING one-hop descent control (see ReferenceChainPassContext:: batch_tags). - jlong my_tag = *tag_ptr; - if (my_tag > 0 && ctx->batch_tags->count(my_tag) != 0) { - // The previously visited batch entry (if any) is now fully processed - record it for the - // order-independent truncated-batch resume (see - // ReferenceChainPassContext::_completed_batch_tags' own comment). - if (ctx->_last_visited_batch_tag != 0 && - ctx->_last_visited_batch_tag != my_tag && - ctx->_completed_batch_tags != nullptr) { - ctx->_completed_batch_tags->insert(ctx->_last_visited_batch_tag); - } - // Track this batch entry as visited for the rolling resume cursor (see - // _last_visited_batch_tag's own comment). - ctx->_last_visited_batch_tag = my_tag; - return JVMTI_VISIT_OBJECTS; - } - return 0; - } - - return JVMTI_VISIT_OBJECTS; -} - -ReferenceChainTracker::AdmitResult ReferenceChainTracker::admitObject( - FrontierTable *frontier, int hop_cap, int budget, int *edges_admitted, - jlong *tag_ptr, jlong parent_tag, u32 referrer_klass, u32 depth, - u8 root_kind, jlong class_tag, bool priority, - jint edge_field_index, u8 edge_kind, jlong edge_referrer_class_tag) { - // edge_* default-declared in the header; heapRootCallback() passes the defaults (a root reference - // is not a field edge) unchanged. - if (*tag_ptr != 0) { - return AdmitResult::ALREADY_ADMITTED; - } - if (depth >= (u32)hop_cap) { - return AdmitResult::HOP_CAP; - } - if (*edges_admitted >= budget) { - return AdmitResult::BUDGET_EXHAUSTED; - } - jlong tag = nextTag(); - if (!frontier->insert(tag, parent_tag, referrer_klass, depth, - FrontierEntryState::FRONTIER, root_kind, class_tag, - edge_field_index, edge_kind, edge_referrer_class_tag)) { - return AdmitResult::FRONTIER_CAP_HIT; - } - *tag_ptr = tag; - (*edges_admitted)++; - // Queue for expandFrontier()/markAllFrontierExpanded() - see _pending_expand's/_priority_expand's - // own declaration comments for why this replaces a scan over the admitted range, and for why a - // rotation-discovered child (priority=true) skips the ordinary backlog. - if (priority && _priority_expand.size() < PRIORITY_EXPAND_CAP) { - _priority_expand.push_back(tag); - _priority_expand_set.insert(tag); - } else { - // Priority lane full: the rotation backpressure falls back to the ordinary backlog rather than - // silently dropping the re-discovered subtree (see PRIORITY_EXPAND_CAP's own comment). - _pending_expand.push_back(tag); - } - trackLeakAccumulation(frontier, class_tag, parent_tag, tag); - return AdmitResult::ADMITTED; -} - -void ReferenceChainTracker::trackLeakAccumulation(FrontierTable *frontier, - jlong class_tag, - jlong parent_tag, - jlong tag) { - // Cheapest checks first: no klass_id is currently watched (the common case before hasLeakSignal() - // has ever fired - see _watched_leak_klass_ids' own comment), or this admission has no real - // parent to attribute to (a root-attached entry - nothing to aggregate by, since the "container" - // concept this tracks is specifically about a PARENT object's field holding the leaf, not the - // leaf itself being root-attached). - if (_watched_leak_klass_count <= 0 || parent_tag == 0 || class_tag == 0) { - return; - } - // (u32) truncation matches _watched_leak_klass_ids' own storage (see that field's comment) - - // class tags are small, negative, sequentially-minted values in practice - // (ClassTagAllocator::next()), so this never actually loses distinguishing information; it just - // keeps the comparison and the signature-key packing below in the same 32-bit space both already - // used for the (superseded) classMap-id scheme. - u32 truncated_class_tag = (u32)class_tag; - bool watched = false; - for (int i = 0; i < _watched_leak_klass_count; i++) { - if (_watched_leak_klass_ids[i] == truncated_class_tag) { - watched = true; - break; - } - } - if (!watched) { - return; - } - FrontierEntry parent_entry{}; - if (!frontier->lookup(parent_tag, &parent_entry) || - parent_entry.class_tag == 0) { - // Parent since pruned/dead between its own admission and this child's, or admitted before this - // field existed on it (should not happen in practice - class_tag is set at every admission - - // but a stale/unknown parent identity is not something to attribute this observation to either - // way. - return; - } - u64 key = leakSignatureKey(truncated_class_tag, (u32)parent_entry.class_tag); - _leak_signature_totals[key]++; - auto it = _leak_parent_fanout.find(parent_tag); - if (it == _leak_parent_fanout.end()) { - TEST_LOG("ReferenceChainTracker::trackLeakAccumulation fanout-insert " - "parent_tag=%lld parent_class_tag=%lld child_class_tag=%lld", - (long long)parent_tag, (long long)parent_entry.class_tag, - (long long)class_tag); - _leak_parent_fanout.emplace(parent_tag, LeakParentFanoutEntry{key, 1}); - } else { - // The signature key for a given parent_tag is fixed once recorded (parent_entry.class_tag never - // changes once admitted; the LEAF side of the key is fixed by which klass_id is currently - // watched at the time of THIS call, which could in principle differ between two children of the - // same parent if _watched_leak_klass_ids itself changed between them - overwrite rather than - // accumulate under a stale key in that case, since the stored signature_key should always - // reflect the most recently observed watched klass_id for this parent). - it->second.signature_key = key; - it->second.fanout++; - } - // ANCESTOR FANOUT: the direct parent is not necessarily the part of the holder chain that STAYS - // LIVE. - jlong ancestor = parent_entry.parent_tag; - int hops = 0; - while (ancestor != 0 && hops++ < _hop_cap) { - FrontierEntry ancestor_entry{}; - if (!_frontier->lookup(ancestor, &ancestor_entry)) { - break; - } - if (_leak_parent_fanout.find(ancestor) == _leak_parent_fanout.end()) { - _leak_parent_fanout.emplace(ancestor, LeakParentFanoutEntry{key, 1}); - } - if (ancestor_entry.parent_tag == 0) { - break; // root-attached: the holder chain ends here - } - ancestor = ancestor_entry.parent_tag; - } -} - -void ReferenceChainTracker::seedLeakAccumulationForNewlyWatchedKlass( - u32 klass_id) { - if (_frontier == nullptr) { - // pollWatchedTargets() can run before the first pass has ever created the frontier table - - // nothing to seed from yet. - return; - } - int table_size = _frontier->size(); - if (table_size <= 0) { - return; - } - // Inlines trackLeakAccumulation()'s own signature/fanout update logic (rather than calling it per - // matching entry) deliberately: this whole scan already holds _frontier's shared lock for its - // duration (matching collectStaleExpandedEntriesForRotation()'s own lockShared() rationale - a - // per-tag SpinLock acquisition would double the cost of this O(table_size) sweep), and - // trackLeakAccumulation() takes that same lock itself via frontier->lookup() - calling it from - // inside an already-held shared section would risk a reentrant-lock deadlock if a writer is ever - // concurrently pending, so this uses lookupLocked() throughout instead. - _frontier->withSharedLock([&](const FrontierTable *frontier) { - for (jlong tag = 1; tag <= table_size; tag++) { - FrontierEntry entry{}; - if (!frontier->lookupLocked(tag, &entry) || - entry.state != FrontierEntryState::EXPANDED || - entry.parent_tag == 0 || (u32)entry.class_tag != klass_id) { - continue; - } - FrontierEntry parent_entry{}; - if (!frontier->lookupLocked(entry.parent_tag, &parent_entry) || - parent_entry.class_tag == 0) { - continue; - } - u64 key = leakSignatureKey(klass_id, (u32)parent_entry.class_tag); - _leak_signature_totals[key]++; - auto it = _leak_parent_fanout.find(entry.parent_tag); - if (it == _leak_parent_fanout.end()) { - _leak_parent_fanout.emplace(entry.parent_tag, - LeakParentFanoutEntry{key, 1}); - } else { - it->second.signature_key = key; - it->second.fanout++; - } - } - }); -} - -bool ReferenceChainTracker::maybeUpgradeRootAttachedRootKind( - FrontierTable *frontier, jlong tag, u8 new_root_kind) { - FrontierEntry entry{}; - if (!frontier->lookup(tag, &entry)) { - return false; - } - if (entry.parent_tag != 0) { - // Not root-attached - per this phase's option (a) resolution of the parent_tag==0/root_kind - // invariant conflict (referenceChains.h's FrontierEntry::root_kind comment), only a - // root-context update may ever write a non-zero root_kind, and only onto an entry that is - // already root-attached. - return false; - } - if (rootKindDurability(new_root_kind) <= rootKindDurability(entry.root_kind)) { - return false; - } - frontier->updateRootKind(tag, new_root_kind); - addToStaticAnchorIndex(tag, entry.class_tag, new_root_kind); - return true; -} - -std::vector -ReferenceChainTracker::collectStaleRootKindEntriesForRotation( - int max_count) { - std::vector selected; - int table_size = _frontier->size(); - if (max_count <= 0 || table_size <= 0) { - return selected; - } - if (_root_kind_rotation_cursor <= 0 || - _root_kind_rotation_cursor > table_size) { - _root_kind_rotation_cursor = 1; - } - - // Held for the whole sweep below (potentially wrapping all the way around table_size) rather than - // once per tag via lookup() - the same rationale as collectStaleExpandedEntriesForRotation()'s - // own lockShared() use: a per-tag SpinLock acquisition would double this scan's cost under a - // large frontier table. - jlong start_tag = _root_kind_rotation_cursor; - jlong tag = start_tag; - _frontier->withSharedLock([&](const FrontierTable *frontier) { - do { - FrontierEntry entry{}; - if (frontier->lookupLocked(tag, &entry) && - entry.state == FrontierEntryState::EXPANDED && - entry.parent_tag == 0 && isTransientRootKind(entry.root_kind) && - !isQueuedForRotation(tag) && - _priority_expand.size() < PRIORITY_EXPAND_CAP) { - selected.push_back(tag); - _priority_expand.push_back(tag); - _priority_expand_set.insert(tag); - if ((int)selected.size() >= max_count) { - tag = tag % table_size + 1; - break; - } - } - tag = tag % table_size + 1; - } while (tag != start_tag); - }); - - _root_kind_rotation_cursor = tag; - return selected; -} - -std::vector -ReferenceChainTracker::collectStaleExpandedEntriesForRotation( - int max_count) { - std::vector selected; - int table_size = _frontier->size(); - if (max_count <= 0 || table_size <= 0) { - return selected; - } - // LEAK-PARENT PRIORITY, FAIR-SHARED WITH THE BLIND LAP: _leak_parent_fanout knows the EXPANDED - // parents that actually lead to watched leak-klass children - re-walking one of those re-sees its - // current children (improveChain() upgrades children first admitted via a shallower path, - // leak-tag interception for the tagged ones) and catches elements added since its expansion, - // which is exactly the mutation this rotation exists to observe. - if (!_leak_parent_fanout.empty() && - _priority_expand.size() < PRIORITY_EXPAND_CAP) { - int fanout_budget = (max_count + 1) / 2; - size_t fanout_size = _leak_parent_fanout.size(); - u64 skip = _leak_parent_rotation_cursor % fanout_size; - auto it = _leak_parent_fanout.begin(); - while (it != _leak_parent_fanout.end()) { - if ((int)selected.size() >= fanout_budget || - _priority_expand.size() >= PRIORITY_EXPAND_CAP) { - break; - } - if (skip > 0) { - skip--; - ++it; - continue; - } - jlong parent_tag = it->first; - if (isQueuedForRotation(parent_tag)) { - ++it; - continue; - } - FrontierEntry entry{}; - // Dead parent: either the frontier slot is gone entirely, or it was clear()'d (dead object / - // restart wipe) - clear() marks the slot ABANDONED rather than removing it, so both - // conditions must erase (tags are never reused within a search and the fanout is wiped on - // restart, so an ABANDONED parent can never come back to life). - if (!_frontier->lookup(parent_tag, &entry) || - entry.state == FrontierEntryState::ABANDONED) { - it = _leak_parent_fanout.erase(it); - continue; - } - if (entry.state != FrontierEntryState::EXPANDED) { - ++it; - continue; - } - selected.push_back(parent_tag); - _priority_expand.push_back(parent_tag); - _priority_expand_set.insert(parent_tag); - ++it; - } - _leak_parent_rotation_cursor += selected.size() + 1; - if ((int)selected.size() >= max_count) { - // Budget exhausted by the fanout alone (only possible for max_count == 1, where the fanout's - // ceil-half share is the whole budget) - fanout-priority preserved, and the lap below has - // nothing left to do this pass. - return selected; - } - } - if (_stale_expanded_rotation_cursor <= 0 || - _stale_expanded_rotation_cursor > table_size) { - _stale_expanded_rotation_cursor = 1; - } - // Resume scanning from _stale_expanded_rotation_cursor rather than always restarting at tag 1: a - // frontier table can accumulate far more than max_count entries that are EXPANDED and stay that - // way forever (long-lived infrastructure objects - caches, maps, bootstrap classes). - int deadline_check_counter = 0; - jlong start_tag = _stale_expanded_rotation_cursor; - jlong tag = start_tag; - _frontier->withSharedLock([&](const FrontierTable *frontier) { - do { - if (_pass_deadline_ns != 0 && - (++deadline_check_counter & 0xFFF) == 0 && - OS::nanotime() >= _pass_deadline_ns) { - // Ran past this pass's wall-clock share - stop scanning with whatever was already selected - // (possibly none) and resume from here next call. - break; - } - FrontierEntry entry{}; - if (frontier->lookupLocked(tag, &entry) && - entry.state == FrontierEntryState::EXPANDED && - !isQueuedForRotation(tag) && - _priority_expand.size() < PRIORITY_EXPAND_CAP) { - selected.push_back(tag); - _priority_expand.push_back(tag); - _priority_expand_set.insert(tag); - if ((int)selected.size() >= max_count) { - tag = tag % table_size + 1; - break; - } - } - tag = tag % table_size + 1; - } while (tag != start_tag); - }); - _stale_expanded_rotation_cursor = tag; - return selected; -} - -// Select high-fanout parents of classes reported as growing. -std::vector -ReferenceChainTracker::collectLeakAccumulationCandidatesForRotation( - int max_count) { - std::vector selected; - if (max_count <= 0 || _leak_signature_totals.empty()) { - return selected; - } - - // Tier 1: rank signatures by growth since the last pass's snapshot. - u64 winning_key = 0; - bool have_winner = false; - u32 best_delta = 0; - for (const auto &kv : _leak_signature_totals) { - u32 prev = 0; - auto prev_it = _leak_signature_prev_totals.find(kv.first); - if (prev_it != _leak_signature_prev_totals.end()) { - prev = prev_it->second; - } - u32 delta = kv.second > prev ? kv.second - prev : 0; - if (delta > 0 && (!have_winner || delta > best_delta)) { - have_winner = true; - best_delta = delta; - winning_key = kv.first; - } - } - // Roll the snapshot forward for the NEXT pass's comparison regardless of whether this pass found - // a winner - a signature that didn't grow this pass still needs its current total remembered so a - // future pass's delta is computed against the right baseline, not against however many passes ago - // it was last checked. - _leak_signature_prev_totals = _leak_signature_totals; - if (!have_winner) { - // Nothing grew since last pass - nothing to prioritize this tier this time - // (collectStaleExpandedEntriesForRotation()'s unprioritized fallback still covers this - // population eventually). - return selected; - } - - // Tier 2: within the winning signature only, rank concrete parent objects by their own fanout - - // collected first, then partially sorted, since _leak_parent_fanout's total size is what bounds - // this method's cost (not table_size), and is expected to be small (see that map's own comment). - std::vector> candidates; // (parent_tag, fanout) - for (const auto &kv : _leak_parent_fanout) { - if (kv.second.signature_key == winning_key && !isQueuedForRotation(kv.first)) { - FrontierEntry entry{}; - if (_frontier->lookup(kv.first, &entry) && - (entry.state == FrontierEntryState::EXPANDED || - entry.state == FrontierEntryState::FRONTIER)) { - candidates.emplace_back(kv.first, kv.second.fanout); - } - } - } - std::sort(candidates.begin(), candidates.end(), - [](const std::pair &a, const std::pair &b) { - return a.second > b.second; - }); - for (const auto &c : candidates) { - if ((int)selected.size() >= max_count || - _priority_expand.size() >= PRIORITY_EXPAND_CAP) { - break; - } - selected.push_back(c.first); - _priority_expand_set.insert(c.first); - FrontierEntry state_entry{}; - bool is_expanded = _frontier->lookup(c.first, &state_entry) && - state_entry.state == FrontierEntryState::EXPANDED; - TEST_LOG("ReferenceChainTracker::" - "collectLeakAccumulationCandidatesForRotation selected " - "parent_tag=%lld state=%s fanout=%u", - (long long)c.first, is_expanded ? "EXPANDED" : "FRONTIER", - c.second); - } - // Place the whole selection at the head of the priority lane, keeping the fanout ranking order - // (see the FRONTIER-state case in the Tier 2 comment above for why the head and not the tail): - // push_front reverses, so insert back-to-front. - for (auto it = selected.rbegin(); it != selected.rend(); ++it) { - _priority_expand.push_front(*it); - } - return selected; -} - diff --git a/ddprof-lib/src/main/cpp/referenceChains.cpp b/ddprof-lib/src/main/cpp/referenceChains.cpp index fa5c28fa27..d2cf6cac45 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.cpp +++ b/ddprof-lib/src/main/cpp/referenceChains.cpp @@ -4,7 +4,6 @@ */ #include "referenceChains.h" -#include "referenceChainInternal.h" #include "common.h" #include "counters.h" #include "jniHelper.h" @@ -21,18 +20,23 @@ #include #include #include -#include #include #include #include -#include -#include -#include #include +#include #include #include +// --------------------------------------------------------------------------- // Reference-chains debug-log level (see rcDebugLevel.h). Level 0 silent +// (default), 1 lifecycle/summary, 2 full per-object diagnostics. Sources: +// env DD_PROFILING_REFERENCE_CHAINS_DEBUG, overridden at runtime by +// /tmp/ddprof_root/refchains_debug_level, re-checked ~1/s from threadLoop +// (rcDebugLevelRefresh never runs in heap callbacks - they only read the +// cached atomic below). Compiled in all builds (harmless in non-DEBUG: +// nothing calls it, the macros are no-ops) so the gtest binary can test it. +// --------------------------------------------------------------------------- namespace { constexpr const char *kRcDebugLevelEnv = "DD_PROFILING_REFERENCE_CHAINS_DEBUG"; constexpr const char *kRcDebugLevelFile = "/tmp/ddprof_root/refchains_debug_level"; @@ -51,8 +55,9 @@ int rcDebugLevel() { if (lvl >= 0) { return lvl; } - // Lazy one-time env resolve; may fire from a heap callback on the very first log line, which is - // still strictly cheaper than the fprintf the same line performs in a DEBUG build. + // Lazy one-time env resolve; may fire from a heap callback on the very + // first log line, which is still strictly cheaper than the fprintf the + // same line performs in a DEBUG build. lvl = envRcDebugLevel(); g_rc_debug_level.store(lvl, std::memory_order_relaxed); return lvl; @@ -84,43 +89,39 @@ int readRcDebugLevelFile(const char *path) { if (path == nullptr) { return -1; } - // The knob file lives under the world-writable /tmp (see kRcDebugLevelFile's - // comment): refuse anything that is not a regular file owned by root or the - // current user, so a local user cannot plant a symlink or a pre-created - // file of their own and force the DEBUG-build diagnostics on. The worst - // impact of a forged file is log-volume/CPU from enabled TEST_LOG in a - // DEBUG build, but the check is cheap and keeps the knob owner-scoped. - // Open first (O_NOFOLLOW rejects a symlink swap outright) and fstat the - // resulting descriptor: a stat-then-open sequence would re-resolve the path - // after the check, leaving a TOCTOU window where the checked file is - // swapped for another one before the read. - int fd = open(path, O_RDONLY | O_NOFOLLOW | O_CLOEXEC); - if (fd < 0) { + // Path hardening: the file lives in a shared, world-writable directory + // (/tmp/ddprof_root/), so open it O_NOFOLLOW and verify it is a regular + // file owned by this process's uid before trusting its contents. Without + // this, any local user on a multi-tenant host could plant a symlink (or + // a fifo - fopen blocks on one) and flip the profiler's diagnostics + // verbosity from 0 to 2 (full per-object heap diagnostics on stdout). + // DEBUG-only diagnostics gate, so the failure mode is informational + // degradation, not security exposure of data - but the check is cheap. +#ifdef O_NOFOLLOW + int fd = open(path, O_RDONLY | O_NOFOLLOW); + if (fd == -1) { return -1; } struct stat st; if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || - (st.st_uid != 0 && st.st_uid != geteuid())) { + st.st_uid != geteuid()) { close(fd); return -1; } - char buf[16]; - size_t n = 0; - while (n < sizeof(buf) - 1) { - ssize_t r = read(fd, buf + n, sizeof(buf) - 1 - n); - if (r < 0) { - if (errno == EINTR) { - continue; - } - close(fd); - return -1; - } - if (r == 0) { - break; - } - n += static_cast(r); + FILE *f = fdopen(fd, "r"); + if (f == nullptr) { + close(fd); + return -1; + } +#else + FILE *f = fopen(path, "r"); + if (f == nullptr) { + return -1; } - close(fd); +#endif + char buf[16]; + size_t n = fread(buf, 1, sizeof(buf) - 1, f); + fclose(f); buf[n] = '\0'; return parseRcDebugLevel(buf); } @@ -140,11 +141,457 @@ void rcDebugLevelRefresh(bool force) { g_rc_debug_level.store(lvl, std::memory_order_relaxed); } +// --------------------------------------------------------------------------- +// FrontierTable (tag-indexed frontier metadata table) +// --------------------------------------------------------------------------- + +FrontierTable::FrontierTable(int max_cap) + : _table_size(0), _table_cap(0), _table_max_cap(std::max(max_cap, 0)), + _table(nullptr) { + _table_cap = std::min(INITIAL_TABLE_CAPACITY, _table_max_cap); + if (_table_cap > 0) { + _table = (FrontierEntry *)calloc(_table_cap, sizeof(FrontierEntry)); + if (_table == nullptr) { + _table_cap = 0; + } + } + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + (jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, _table_cap); +} + +FrontierTable::~FrontierTable() { + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + -(jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, -_table_cap); + free(_table); +} + +void FrontierTable::resetCapacityForTest(int max_cap) { + _table_lock.lock(); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + -(jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, -_table_cap); + free(_table); + _table = nullptr; + _table_max_cap = std::max(max_cap, 0); + _table_cap = std::min(INITIAL_TABLE_CAPACITY, _table_max_cap); + if (_table_cap > 0) { + _table = (FrontierEntry *)calloc(_table_cap, sizeof(FrontierEntry)); + if (_table == nullptr) { + _table_cap = 0; + } + } + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + (jlong)_table_cap * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, _table_cap); + _table_size.store(0, std::memory_order_relaxed); + _table_lock.unlock(); +} + +bool FrontierTable::growLocked(int required_cap) { + if (required_cap <= _table_cap) { + return true; + } + if (_table_cap >= _table_max_cap) { + return false; + } + + int newcap = _table_cap; + while (newcap < required_cap && newcap < _table_max_cap) { + newcap = newcap == 0 ? std::min(INITIAL_TABLE_CAPACITY, _table_max_cap) + : std::min(newcap * 2, _table_max_cap); + } + if (newcap <= _table_cap) { + return false; + } + + FrontierEntry *tmp = + (FrontierEntry *)realloc(_table, sizeof(FrontierEntry) * newcap); + if (tmp == nullptr) { + Log::debug( + "ReferenceChains: frontier table resize to %d entries failed", newcap); + return false; + } + // realloc() does not zero the newly grown region - clear it so lookup() + // never returns garbage state for a slot that hasn't been inserted yet. + memset(tmp + _table_cap, 0, sizeof(FrontierEntry) * (newcap - _table_cap)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_BYTES, + (jlong)(newcap - _table_cap) * sizeof(FrontierEntry)); + Counters::increment(REFERENCE_CHAIN_FRONTIER_TABLE_CAPACITY, + newcap - _table_cap); + _table = tmp; + _table_cap = newcap; + return _table_cap >= required_cap; +} + +bool FrontierTable::insert(jlong tag, jlong parent_tag, u32 referrer_klass, + u32 depth, u8 state, u8 root_kind, + jlong class_tag, jint referrer_field_index, + u8 edge_kind, jlong referrer_class_tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + int idx = (int)(tag - 1); + + // Exclusive lock for the whole write (growLocked() already requires it) - + // a shared lock here would not exclude lookup()'s own shared-mode read of + // the same slot, letting a concurrent reader observe a torn entry. + _table_lock.lock(); + if (idx >= _table_cap && !growLocked(idx + 1)) { + _table_lock.unlock(); + Log::debug("ReferenceChains: frontier table capacity exhausted " + "(cap=%d, max=%d, tag=%lld)", + _table_cap, _table_max_cap, (long long)tag); + return false; + } + _table[idx].parent_tag = parent_tag; + _table[idx].referrer_klass = referrer_klass; + _table[idx].depth = depth; + _table[idx].state = state; + _table[idx].root_kind = root_kind; + _table[idx].class_tag = class_tag; + _table[idx].leak_tag = 0; + _table[idx].referrer_field_index = referrer_field_index; + _table[idx].edge_kind = edge_kind; + _table[idx].referrer_class_tag = referrer_class_tag; + _table_lock.unlock(); + + int sz = _table_size.load(std::memory_order_relaxed); + while (sz < idx + 1 && + !_table_size.compare_exchange_weak(sz, idx + 1, + std::memory_order_relaxed)) { + // sz reloaded with the current value by compare_exchange_weak on + // failure; retry until either this thread wins or another thread + // already advanced _table_size past idx + 1. + } + return true; +} + +bool FrontierTable::lookup(jlong tag, FrontierEntry *out) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + int idx = (int)(tag - 1); + + bool found = false; + _table_lock.lockShared(); + if (idx < _table_size) { + *out = _table[idx]; + found = true; + } + _table_lock.unlockShared(); + return found; +} + +bool FrontierTable::lookupLocked(jlong tag, FrontierEntry *out) const { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + int idx = (int)(tag - 1); + if (idx < _table_size) { + *out = _table[idx]; + return true; + } + return false; +} + +void FrontierTable::clear(jlong tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + // Exclusive lock: this mutates a slot lookup() may be reading concurrently + // under its own shared lock (see insert()'s own comment above). + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].state = FrontierEntryState::ABANDONED; + } + _table_lock.unlock(); +} + +void FrontierTable::markEdge(jlong tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].state = FrontierEntryState::EDGE; + } + _table_lock.unlock(); +} + +void FrontierTable::markExpanded(jlong tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].state = FrontierEntryState::EXPANDED; + } + _table_lock.unlock(); +} + +void FrontierTable::updateRootKind(jlong tag, u8 root_kind) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].root_kind = root_kind; + } + _table_lock.unlock(); +} + +bool FrontierTable::improveChain(jlong tag, jlong parent_tag, + u32 referrer_klass, u32 depth, + u8 root_kind, jint referrer_field_index, + u8 edge_kind, jlong referrer_class_tag) { + // Replace a shallow root-attached entry (parent_tag == 0, depth == 0) + // with a deeper chain-attached entry when the object is reached via a + // longer path. This fixes the "depth=1 chain with no holder" problem: + // an object first admitted as a JNI-local root (parent_tag == 0) gets + // its frontier entry overwritten when the static-field → ... → object + // path reaches it later with a non-zero parent_tag. + // Returns true if the entry was actually improved (new depth > old). + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return false; + } + // Round 16 (pod round-15 measurement, ev-leaktag-onpod-round15-results): + // a "chain" whose parent is the entry itself is never an improvement - + // it is the self-edge a this-field produces, and it is REAL in the heap: + // every java.util.Collections$Synchronized* holder carries mutex == this, + // so walking such a holder's own subtree (the rotation anchor walk or a + // BFS descent) re-reports the holder as its own child through that field. + // On the pod this exact edge demoted the LEAK_BUFFER wrapper: admitted + // root-attached (parent=0, root_kind=STATIC_FIELD) in search #2 and + // walked once, then every later search's entry read parent==its own tag + // root_kind=0 referrer_klass= - collector-invisible + // (the parent_tag==0 eligibility filter) and never re-walked, because + // improveChain(depth=holder.depth+1) "improved" the entry with itself. + // A self-parent would also dead-loop reconstructChain(). Refuse. + if (parent_tag == tag) { + return false; + } + // Ancestor-walk bound for the cycle guard below - see its comment. 16x + // the largest hop cap: a legitimate parent chain never comes close. + static constexpr int IMPROVE_CHAIN_GUARD_MAX_HOPS = 4096; + { + int guard_hops = 0; + jlong cur = parent_tag; + while (cur > 0 && guard_hops <= IMPROVE_CHAIN_GUARD_MAX_HOPS) { + if (cur == tag) { + TEST_LOG_SUMMARY("FrontierTable::improveChain refused: new parent " + "chain routes through the entry (cycle) tag=%lld " + "parent_tag=%lld depth=%u", + (long long)tag, (long long)parent_tag, depth); + return false; + } + FrontierEntry guard_entry{}; + if (!lookup(cur, &guard_entry)) { + break; + } + cur = guard_entry.parent_tag; + guard_hops++; + } + if (cur != 0) { + // The parent chain neither reached a root nor was fully verified + // within the guard bound - applying this improve could embed an + // unresolvable (cyclic or dangling) chain. Keep the existing entry. + TEST_LOG_SUMMARY("FrontierTable::improveChain refused: unverifiable " + "parent chain tag=%lld parent_tag=%lld depth=%u " + "walk_stopped_at=%lld", + (long long)tag, (long long)parent_tag, depth, + (long long)cur); + return false; + } + } + + int idx = (int)(tag - 1); + + _table_lock.lock(); + bool improved = false; + if (idx < _table_size && depth > _table[idx].depth) { + _table[idx].parent_tag = parent_tag; + _table[idx].referrer_klass = referrer_klass; + _table[idx].depth = depth; + _table[idx].root_kind = root_kind; + _table[idx].referrer_field_index = referrer_field_index; + _table[idx].edge_kind = edge_kind; + _table[idx].referrer_class_tag = referrer_class_tag; + improved = true; + } + _table_lock.unlock(); + return improved; +} + +bool FrontierTable::reparentToDurableRoot(jlong tag, jlong new_parent_tag, + u32 referrer_klass, + jint referrer_field_index, + u8 edge_kind) { + // See the declaration's own comment (referenceChains.h) for why this + // exists as a sibling of improveChain(): equal-depth depth-1 noise->real + // re-parenting. All lookups happen under one lock - three index reads, + // no allocation, O(1). + // The parent==tag self-edge guard mirrors improveChain's (round 16); + // today this cannot self-reparent - the swap below requires the entry's + // own parent_tag > 0 while the new parent slot (the same slot) must hold + // parent_tag == 0, a contradiction - but the two siblings' contracts + // stay identical so a future caller change cannot reintroduce the + // self-parent a this-field (mutex == this) would deliver. + if (tag <= 0 || tag - 1 > (jlong)INT_MAX || new_parent_tag <= 0 || + new_parent_tag - 1 > (jlong)INT_MAX || new_parent_tag == tag) { + return false; + } + int idx = (int)(tag - 1); + int new_par_idx = (int)(new_parent_tag - 1); + + _table_lock.lock(); + bool swapped = false; + if (idx < _table_size && _table[idx].depth == 1 && + _table[idx].parent_tag > 0 && _table[idx].parent_tag != new_parent_tag) { + int old_par_idx = (int)(_table[idx].parent_tag - 1); + if (old_par_idx >= 0 && old_par_idx < _table_size && + new_par_idx < _table_size && + _table[new_par_idx].parent_tag == 0 && + _table[new_par_idx].root_kind != 0 && + !isTransientRootKind(_table[new_par_idx].root_kind) && + _table[old_par_idx].parent_tag == 0 && + isTransientRootKind(_table[old_par_idx].root_kind)) { + // New parent is a root-attached DURABLE root (static field, JNI + // global, thread) and the current parent is a root-attached TRANSIENT + // one - same depth, strictly better retention explanation. + _table[idx].parent_tag = new_parent_tag; + _table[idx].referrer_klass = referrer_klass; + _table[idx].referrer_field_index = referrer_field_index; + _table[idx].edge_kind = edge_kind; + swapped = true; + } + } + _table_lock.unlock(); + return swapped; +} + +bool FrontierTable::reconstructChain(jlong target_tag, + std::vector *out_chain, + u8 *out_root_kind, + std::vector *out_edges, + FrontierEntry *out_terminal) { + FrontierEntry entry{}; + if (!lookup(target_tag, &entry)) { + return false; + } + + std::vector chain; + std::vector edges; + jlong tag = target_tag; + u8 root_kind = 0; + int hops = 0; + // Bounded by maxCapacity(): every tag maps to a distinct slot (this table's + // "tags/slots are never reused" invariant, see the class comment above), + // so a well-formed parent_tag chain can visit at most maxCapacity() slots + // before either reaching parent_tag == 0 or repeating a slot. + for (; hops <= maxCapacity() && tag != 0; hops++) { + if (!lookup(tag, &entry)) { + // parent_tag pointed at a tag that was never inserted - should not + // happen for a chain built entirely within one BFS pass, but do not + // fabricate a partial chain silently. + TEST_LOG_SUMMARY("FrontierTable::reconstructChain broken chain: " + "target=%lld failed at hop=%d tag=%lld (parent tag never " + "inserted)", + (long long)target_tag, hops, (long long)tag); + return false; + } + chain.push_back(entry.referrer_klass); + if (out_edges != nullptr) { + // edges[i] describes the edge INTO chain[i]: the entry's own recorded + // edge identity, plus the referrer's class tag - the parent entry's + // own class for interior hops, the declaring class for root-attached + // static edges (FrontierEntry::referrer_class_tag, filled only there, + // since a class-object referrer has no parent entry to read from). + ChainHopEdge hop{}; + hop.field_index = entry.referrer_field_index; + if (entry.parent_tag == 0) { + hop.edge_kind = entry.root_kind; + hop.referrer_class_tag = entry.referrer_class_tag; + } else { + hop.edge_kind = entry.edge_kind; + FrontierEntry parent_entry{}; + hop.referrer_class_tag = + lookup(entry.parent_tag, &parent_entry) ? parent_entry.class_tag : 0; + } + edges.push_back(hop); + } + markEdge(tag); + root_kind = entry.root_kind; + tag = entry.parent_tag; + } + if (tag != 0) { + // Ran past the defensive hop bound without reaching a root-attached + // entry (parent_tag == 0) - a corrupted/cyclic chain. Report failure + // rather than returning a truncated, possibly-misleading chain. The + // tag->parent dump names the cycle members (improveChain()'s cycle + // guard keeps new ones from forming, but a cycle written before that + // guard existed - or a dangling parent from a concurrent restart - + // still lands here). + { + jlong dbg = target_tag; + FrontierEntry dbg_e{}; + char pairs[256]; + size_t off = 0; + for (int d = 0; d < 12 && dbg != 0 && off < sizeof(pairs) - 24; d++) { + if (!lookup(dbg, &dbg_e)) { + break; + } + off += (size_t)snprintf(pairs + off, sizeof(pairs) - off, "%lld->%lld ", + (long long)dbg, (long long)dbg_e.parent_tag); + dbg = dbg_e.parent_tag; + } + TEST_LOG_SUMMARY("FrontierTable::reconstructChain hop bound: " + "target=%lld stuck at tag=%lld after %d hops - cyclic or " + "corrupt parent chain; hops: %.*s", + (long long)target_tag, (long long)tag, hops, (int)off, pairs); + } + return false; + } + + *out_chain = std::move(chain); + if (out_edges != nullptr) { + *out_edges = std::move(edges); + } + if (out_root_kind != nullptr) { + // The loop's last iteration is always the root-attached entry (the one + // whose parent_tag == 0 that just ended the loop), so root_kind here is + // that entry's own FrontierEntry::root_kind. + *out_root_kind = root_kind; + } + if (out_terminal != nullptr) { + // `entry` still holds the loop's last successful lookup - the + // root-attached entry that ended the walk. + *out_terminal = entry; + } + return true; +} + +// --------------------------------------------------------------------------- // ReferenceChainTracker +// --------------------------------------------------------------------------- -// Marks the calling thread as executing inside the GarbageCollectionStart/ Finish JVMTI callback -// for the duration of the guard's lifetime. -thread_local bool t_inGCCallback = false; +// Marks the calling thread as executing inside the GarbageCollectionStart/ +// Finish JVMTI callback for the duration of the guard's lifetime. Used by the +// tag helpers below as a debug-only self-consistency check that this class +// never issues a Heap-category JVMTI call (SetTag/GetTag/...) from a context +// where the JVMTI spec forbids it (see referenceChains.h). Thread-local +// because the JVMTI spec only guarantees the callback runs on the VM thread +// delivering the event, and this must not leak across threads. +static thread_local bool t_inGCCallback = false; namespace { class GCCallbackGuard { @@ -158,16 +605,16 @@ void ReferenceChainTracker::autoTuneDefaults(Arguments &args) { // Only tune defaults the operator did not set explicitly. const u8 tuned = args._reference_chains_tuned_mask; - // Max heap is resolved by LivenessTracker::initialize_table() at this point - // (ObjectSampler::start() -> LivenessTracker::start() runs before ReferenceChainTracker::start() - // in Profiler::start()). + // Max heap is resolved by LivenessTracker::initialize_table() at this + // point (ObjectSampler::start() -> LivenessTracker::start() runs + // before ReferenceChainTracker::start() in Profiler::start()). jlong max_heap = LivenessTracker::instance()->maxHeapBytes(); if (max_heap <= 0) { return; // can't tune without heap size } - // Available processors from JVMTI (cached by FlightRecorder, but we can query JVMTI directly - // here). + // Available processors from JVMTI (cached by FlightRecorder, but we + // can query JVMTI directly here). jint nprocs = 1; jvmtiEnv *jvmti = VM::jvmti(); if (jvmti != nullptr) { @@ -178,24 +625,33 @@ void ReferenceChainTracker::autoTuneDefaults(Arguments &args) { // Heap size in MiB. double heap_mib = (double)max_heap / (1024.0 * 1024.0); - // --- Budget (edges per BFS pass) --- Scale with sqrt(heap_mib): a 4 GiB heap gets 2x, a 16 GiB - // heap gets 4x, a 64 GiB heap gets 8x the default 1000. + // --- Budget (edges per BFS pass) --- + // Scale with sqrt(heap_mib): a 4 GiB heap gets 2x, a 16 GiB heap + // gets 4x, a 64 GiB heap gets 8x the default 1000. This keeps the + // per-pass safepoint pause proportional to sqrt(heap) — the + // number of edges explored per pass grows, but not quadratically. if (!(tuned & REF_CHAINS_TUNED_BUDGET)) { int scaled = (int)(DEFAULT_REFERENCE_CHAINS_BUDGET * std::sqrt(heap_mib / 512.0)); args._reference_chains_budget = std::max(DEFAULT_REFERENCE_CHAINS_BUDGET, std::min(scaled, MAX_REFERENCE_CHAINS_BUDGET)); } - // --- First-pass budget --- The root enumeration pass is one-shot per search and can afford a - // much larger budget. + // --- First-pass budget --- + // The root enumeration pass is one-shot per search and can + // afford a much larger budget. Scale it 10x the per-pass budget + // so the first pass covers more roots. if (!(tuned & REF_CHAINS_TUNED_FIRST_PASS_BUDGET)) { int fpb = args._reference_chains_budget * 10; args._reference_chains_first_pass_budget = std::min(fpb, MAX_REFERENCE_CHAINS_FIRST_PASS_BUDGET); } - // --- TTL (per-search wall-clock lifetime) --- The search needs enough time to cover the heap at - // the tuned budget. + // --- TTL (per-search wall-clock lifetime) --- + // The search needs enough time to cover the heap at the tuned + // budget. At ~1 pass/sec, TTL_seconds >= heap_edges / budget. + // We don't know heap_edges, but it scales with heap size. Use + // heap_mib as a proxy: TTL = base_ttl * (heap_mib / 512), + // clamped to [60s, 30min]. if (!(tuned & REF_CHAINS_TUNED_TTL)) { long scaled_ttl = (long)(DEFAULT_REFERENCE_CHAINS_TTL_MS * (heap_mib / 512.0)); scaled_ttl = std::max(DEFAULT_REFERENCE_CHAINS_TTL_MS, std::min(scaled_ttl, @@ -203,7 +659,13 @@ void ReferenceChainTracker::autoTuneDefaults(Arguments &args) { args._reference_chains_ttl_ms = scaled_ttl; } - // --- Frontier cap --- The frontier grows with the number of edges admitted per pass. + // --- Frontier cap --- + // The frontier grows with the number of edges admitted per + // pass. Scale with budget so a larger budget doesn't + // immediately hit the cap. Use a floating-point ratio - integer + // division here would truncate the scale factor (e.g. a budget + // of 3741 against a default of 1000 would floor to a 3x + // multiplier instead of ~3.74x, undershooting the cap by ~20%). if (!(tuned & REF_CHAINS_TUNED_FRONTIER_CAP)) { int scaled_cap = (int)(DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP * ((double)args._reference_chains_budget / DEFAULT_REFERENCE_CHAINS_BUDGET)); @@ -212,15 +674,21 @@ void ReferenceChainTracker::autoTuneDefaults(Arguments &args) { std::min(scaled_cap, MAX_REFERENCE_CHAINS_FRONTIER_CAP)); } - // --- Pause target --- More available processors = the JVM can afford a slightly longer per-pass - // safepoint without impacting application throughput. + // --- Pause target --- + // More available processors = the JVM can afford a slightly + // longer per-pass safepoint without impacting application + // throughput. Scale linearly: 1 core = 50ms, 4 cores = 100ms, + // 8 cores = 150ms, capped at 50ms (the per-call STW cap from the + // safepoint budget model). if (!(tuned & REF_CHAINS_TUNED_PAUSE_TARGET)) { long scaled_pause = DEFAULT_REFERENCE_CHAINS_PAUSE_TARGET_MS * (1 + (nprocs - 1) / 3); args._reference_chains_pause_target_ms = std::min(scaled_pause, (long)50); } - // --- Pain budget percent --- More cores = more spare capacity for background work. + // --- Pain budget percent --- + // More cores = more spare capacity for background work. + // Scale: 1 core = 1%, 4 cores = 2%, 8 cores = 3%, capped at 5%. if (!(tuned & REF_CHAINS_TUNED_PAIN_BUDGET)) { int scaled_pain = DEFAULT_REFERENCE_CHAINS_PAIN_BUDGET_PERCENT * (1 + (nprocs - 1) / 4); @@ -245,11 +713,14 @@ Error ReferenceChainTracker::start(Arguments &args) { return Error::OK; } - // Recording-boundary hygiene: Profiler::start() clears the class dictionary (restart its id - // namespace) right before this runs, so cached chain events and queued abandonment events from a - // prior recording carry StringDictionary ids from a wiped generation - re-emitting them into the - // new recording would write missing or newly-reassigned class ids for chains that describe the - // previous recording's objects. + // Recording-boundary hygiene: Profiler::start() clears the class dictionary + // (restart its id namespace) right before this runs, so cached chain events + // and queued abandonment events from a prior recording carry StringDictionary + // ids from a wiped generation - re-emitting them into the new recording would + // write missing or newly-reassigned class ids for chains that describe the + // previous recording's objects. The frontier table and class-tag cache are + // deliberately KEPT across recordings (their own comments); the chain-event + // caches are not - they are pure recording output. _resolved_chains_lock.lock(); _resolved_chains.clear(); _resolved_chains_lock.unlock(); @@ -258,8 +729,9 @@ Error ReferenceChainTracker::start(Arguments &args) { _pending_abandoned_events_lock.unlock(); _urgency_budget_boosted = false; - // Auto-tune defaults that the operator did not set explicitly, based on max heap size and - // available processors. + // Auto-tune defaults that the operator did not set explicitly, + // based on max heap size and available processors. Must run before + // _configured_frontier_cap is read below. autoTuneDefaults(args); Log::info("Reference chain tracking is enabled (hops=%d, budget=%d, " @@ -269,31 +741,44 @@ Error ReferenceChainTracker::start(Arguments &args) { args._reference_chains_pause_target_ms, args._reference_chains_pain_budget_percent); - // Like LivenessTracker's own table, construct the frontier table once and keep it across repeated - // start()/stop() cycles - do not reallocate on a second start() with a possibly different cap, - // for the same reason LivenessTracker keeps its first-initialize() result. + // Like LivenessTracker's own table, construct the + // frontier table once and keep it across repeated start()/stop() cycles - + // do not reallocate on a second start() with a possibly different cap, for + // the same reason LivenessTracker keeps its first-initialize() result. + // Recorded unconditionally, even on a start() call that finds _frontier + // already constructed (see _configured_frontier_cap's own comment) - this + // is what resetSearchStateForTest() rebuilds the table at, undoing + // whatever cap an earlier test in this same JVM happened to construct it + // with. _configured_frontier_cap = args._reference_chains_frontier_cap; if (_frontier == nullptr) { _frontier = new FrontierTable(_configured_frontier_cap); } - // The configured budget is what the urgency ramp restores when urgency clears (see the urgency - // block in threadLoop()) - the live _budget must not be snapshotted for that, it may already be - // boosted. + // The configured budget is what the urgency ramp restores when urgency + // clears (see the urgency block in threadLoop()) - the live _budget must + // not be snapshotted for that, it may already be boosted. _configured_budget = args._reference_chains_budget; _hop_cap = args._reference_chains_hop_cap; _budget = args._reference_chains_budget; - // 0 (unset) auto-scales from _budget instead of falling back to it plainly - see this field's own - // comment (referenceChains.h) for why a steady-state per-pass budget is the wrong size for the - // first pass. + // 0 (unset) auto-scales from _budget instead of falling back to it plainly + // - see this field's own comment (referenceChains.h) for why a + // steady-state per-pass budget is the wrong size for the first pass. _first_pass_budget = args._reference_chains_first_pass_budget > 0 ? args._reference_chains_first_pass_budget : std::min(_budget * AUTO_FIRST_PASS_BUDGET_MULTIPLIER, AUTO_FIRST_PASS_BUDGET_CAP); _ttl_ms = args._reference_chains_ttl_ms; - // Pause-time pacing controller: (re)seed the controller's ceiling and the adaptive values it - // drives. + // Pause-time pacing controller: (re)seed the controller's ceiling and the + // adaptive values it drives. _effective_budget/_effective_cadence_ns start + // exactly at their pre-pacing-controller fixed-constant equivalents + // (_budget/PASS_CADENCE_NS) so a tracker that has not yet measured a pass + // behaves identically to before the controller was added - updatePacing() + // only moves them once a real pass duration is + // available. _pause_pid is reconstructed (not just reset()) because its + // target is only known now, from args - same reason RateLimiter::start() + // reconstructs its own _pid rather than mutating it in place. _pause_target_ms = args._reference_chains_pause_target_ms; _effective_pause_target_ms = _pause_target_ms; _effective_budget = _budget; @@ -305,65 +790,92 @@ Error ReferenceChainTracker::start(Arguments &args) { sizeof(_candidate_qualifying_tid_count)); _passes_since_last_candidate_progress = 0; _last_candidate_progress_mark = 0; - // Fresh chase gets a fresh back-to-back spacing allowance (see _canary_backoff_mult's own - // comment). + // Fresh chase gets a fresh back-to-back spacing allowance (see + // _canary_backoff_mult's own comment). _canary_backoff_mult = 1; - _canary_pass_ema_ms = 0; + _canary_pass_ema_ns = 0; _last_canary_pass_ns = 0; _canary_stuck_restart_count = 0; - // Budget-borrowing (referenceChains.h's _borrowed_budget comment): reset alongside the rest of - // the pacing controller's state, so a restarted search never inherits headroom earned by a - // previous one. + // Budget-borrowing (referenceChains.h's _borrowed_budget comment): reset + // alongside the rest of the pacing controller's state, so a restarted + // search never inherits headroom earned by a previous one. _borrowed_budget = 0; _consecutive_under_target_passes = 0; _pause_pid = PidController((u64)std::max(_pause_target_ms, 0L), 10, // proportional gain: reacts to a single - // pass's over/under-ceiling error without needing many passes to - // notice - a duration-ms error is typically single/ - // low-double-digit in magnitude (unlike the shared triple's - // event-count scale), so a smaller P keeps a one-pass overshoot - // from. + // pass's over/under-ceiling error without + // needing many passes to notice - a + // duration-ms error is typically single/ + // low-double-digit in magnitude (unlike + // the shared triple's event-count scale), + // so a smaller P keeps a one-pass + // overshoot from swinging the budget by + // more than a modest fraction of itself 1, // integral gain: small and round - - // pidController.cpp's `_integral_value` has no built-in clamp, - // and this controller is invoked once per BFS pass rather than - // on the other three usages' roughly-periodic - // one-call-per-second cadence, so windup accumulates faster per - // wall-clock. + // pidController.cpp's `_integral_value` + // has no built-in clamp, and this + // controller is invoked once per BFS pass + // rather than on the other three usages' + // roughly-periodic one-call-per-second + // cadence, so windup accumulates faster + // per wall-clock second than it does there 2, // derivative gain: small, matching the - // shared triple's own "the derivational gain is rather small" - // rationale (objectSampler.cpp) - a single slow/ fast pass - // should not itself trigger a large swing + // shared triple's own "the derivational + // gain is rather small" rationale + // (objectSampler.cpp) - a single slow/ + // fast pass should not itself trigger a + // large swing 1, // sampling_window=1: one compute() call - // *is* one pass, not a fixed real-time window like the other - // three usages assume (see _pause_pid's own comment) + // *is* one pass, not a fixed real-time + // window like the other three usages + // assume (see _pause_pid's own comment) 5.0 // cutoff_secs: a round value, halved from - // the shared triple's own "15" since a pass-scoped signal is - // naturally noisier per-call than a roughly-1s- cadence one + // the shared triple's own "15" since a + // pass-scoped signal is naturally + // noisier per-call than a roughly-1s- + // cadence one ); - // (re)seed _safepoint_pain_budget from the configured refill rate, mirroring _pause_pid's own - // reconstruct-in-start() pattern above. + // Search restart (this class's own header comment): (re)seed _safepoint_pain_budget + // from the configured refill rate, mirroring _pause_pid's own + // reconstruct-in-start() pattern above. A search's already-accumulated + // _search_pain_ms is deliberately left untouched here - only restartSearch() + // spends it, so a start()/stop() cycle mid-search (if that ever happens) + // does not erase cost the current search has already incurred. _safepoint_pain_budget = PainBudget( std::max(args._reference_chains_pain_budget_percent, 0) / 100.0); _pain_budget_refill_rate = std::max(args._reference_chains_pain_budget_percent, 0) / 100.0; - // Same refill rate as _safepoint_pain_budget above - one operator-facing "how much background - // cost is acceptable" percentage covers both leaky buckets (see _cpu_pain_budget's own comment, - // referenceChains.h). + // Same refill rate as _safepoint_pain_budget above - one operator-facing + // "how much background cost is acceptable" percentage covers both + // leaky buckets (see _cpu_pain_budget's own comment, referenceChains.h). _cpu_pain_budget = PainBudget(_pain_budget_refill_rate); - // Lazy-enable, matching LivenessTracker::start(): the GC callbacks are wired unconditionally in - // vmEntry.cpp, but the events themselves are only turned on for this JVMTI env when the flag is - // on. + // Lazy-enable, matching LivenessTracker::start(): + // the GC callbacks are wired unconditionally in vmEntry.cpp, but the events + // themselves are only turned on for this JVMTI env when the flag is on. + // Null-guarded: this file's own gtest binary calls start() directly with + // no live JVM (the reason the BFS thread is not created here either - see + // below), and VM::jvmti() is null there. autoTuneDefaults() already + // null-checks the same pointer one configuration step earlier. jvmtiEnv *jvmti = VM::jvmti(); - jvmti->SetEventNotificationMode( - JVMTI_ENABLE, JVMTI_EVENT_GARBAGE_COLLECTION_START, nullptr); - jvmti->SetEventNotificationMode( - JVMTI_ENABLE, JVMTI_EVENT_GARBAGE_COLLECTION_FINISH, nullptr); + if (jvmti != nullptr) { + jvmti->SetEventNotificationMode( + JVMTI_ENABLE, JVMTI_EVENT_GARBAGE_COLLECTION_START, nullptr); + jvmti->SetEventNotificationMode( + JVMTI_ENABLE, JVMTI_EVENT_GARBAGE_COLLECTION_FINISH, nullptr); + } - // Deliberately does NOT create the BFS thread (threadEntry()/threadLoop() below) here - - // threadLoop()'s VM::attachThread() call dereferences VM::_vm unconditionally (vmEntry.h:191-195) - // and crashes if the VM is not yet attached, which is exactly the case in this file's own gtest - // binary (referenceChains_ut.cpp calls start() directly with no live JVM). + // Deliberately does NOT create the BFS thread (threadEntry()/threadLoop() + // below) here - threadLoop()'s VM::attachThread() call dereferences + // VM::_vm unconditionally (vmEntry.h:191-195) and crashes if the VM is not + // yet attached, which is exactly the case in this file's own gtest binary + // (referenceChains_ut.cpp calls start() directly with no live JVM). + // startThread() (referenceChains.h) owns spawning the thread instead, and + // is called from Profiler::start() (profiler.cpp) immediately after this + // method returns Error::OK - by that point in the real profiler lifecycle + // the JVM/JVMTI environment is already fully up, so VM::attachThread() is + // safe there. runPass() - the actual BFS engine - does not depend on the + // thread either way and is called directly by this file's own tests. return Error::OK; } @@ -374,22 +886,41 @@ void ReferenceChainTracker::stop() { } Log::info("Reference chain tracking stopped"); - // Do not disable GC notifications here - LivenessTracker follows the same rule since the JVMTI - // env and its tracker singletons are expected to survive across multiple start/stop recording - // cycles. + // Do not disable GC notifications here - LivenessTracker follows the same + // rule since the JVMTI env and its tracker + // singletons are expected to survive across multiple start/stop recording + // cycles. The BFS thread itself is stopped separately, by + // Profiler::stop() calling stopThread() (profiler.cpp) - mirroring + // start()'s split between this method and startThread(). } void ReferenceChainTracker::startThread() { if (!_enabled || _running.load(std::memory_order_acquire)) { return; } - // Reset from any previous stopThread() call - a dynamic-attach profiler can go through multiple - // start()/stop() cycles in one JVM lifetime (this class's own start()/stop() header comments), - // and a stale abort request left set from the prior cycle would make heapReferenceCallback() - // abort this new cycle's very first pass instantly. + // Reset from any previous stopThread() call - a dynamic-attach profiler + // can go through multiple start()/stop() cycles in one JVM lifetime (this + // class's own start()/stop() header comments), and a stale abort request + // left set from the prior cycle would make heapReferenceCallback() abort + // this new cycle's very first pass instantly. _abort_pass_requested.store(false, std::memory_order_relaxed); - // Publish _running=true *before* creating the thread, not after. + // Publish _running=true *before* creating the thread, not after. If the + // OS schedules the new thread ahead of the parent, threadLoop()'s startup + // check (`while (_running.load(...))`) would otherwise be racing against + // this store: the child could see the still-`false` initial value, fall + // straight through the loop, detach and exit - and the parent would then + // publish `true` regardless, leaving startThread() reporting the tracker + // as running while no BFS thread is actually alive for the rest of the + // recording. pthread_create() itself is the fix's synchronization point: + // POSIX guarantees everything the calling thread writes before this call + // is visible to the new thread once it starts running, so ordering the + // store first removes the race outright rather than narrowing it. Roll + // back on a failed create so a later startThread() call is not blocked by + // a stale `_running=true` with no thread behind it. stopThread() is only + // ever called after this method has returned (Profiler::start()/stop() + // pair the two sequentially - see this class's own start()/stop() header + // comments), so its use of _thread below is unaffected by this reordering. _running.store(true, std::memory_order_release); pthread_t thread; if (pthread_create(&thread, NULL, threadEntry, this) != 0) { @@ -405,15 +936,17 @@ void ReferenceChainTracker::stopThread() { return; } _running.store(false, std::memory_order_release); - // Ask any in-flight JVMTI FollowReferences walk (heapReferenceCallback()) to abort at its next - // callback invocation - set before pthread_kill() below, since that signal alone cannot interrupt - // a call already inside the JVM/JVMTI implementation. + // Ask any in-flight JVMTI FollowReferences walk (heapReferenceCallback()) + // to abort at its next callback invocation - set before pthread_kill() + // below, since that signal alone cannot interrupt a call already inside + // the JVM/JVMTI implementation. _abort_pass_requested.store(true, std::memory_order_relaxed); // Same wake-then-join shape as BaseWallClock::stop() (wallClock.cpp:324-333): - // pthread_kill(WAKEUP_SIGNAL) interrupts threadLoop()'s OS::sleep() early (WAKEUP_SIGNAL/SIGIO is - // installed with a no-op handler unconditionally in vmEntry.cpp, so this signal never terminates - // the thread) so it re-checks _running and exits promptly rather than waiting out the rest of the - // current sleep interval. + // pthread_kill(WAKEUP_SIGNAL) interrupts threadLoop()'s OS::sleep() early + // (WAKEUP_SIGNAL/SIGIO is installed with a no-op handler unconditionally + // in vmEntry.cpp, so this signal never terminates the thread) so it + // re-checks _running and exits promptly rather than waiting out the rest + // of the current sleep interval. pthread_kill(_thread, WAKEUP_SIGNAL); int res = pthread_join(_thread, NULL); if (res != 0) { @@ -421,28 +954,50 @@ void ReferenceChainTracker::stopThread() { } } -// Runs scheduled passes on an attached agent thread. +// Not yet started by anything (see start()'s comment above for why) - but +// now implements the real scheduling loop the design doc asks for, matching +// J9WallClock's attach/park/detach lifecycle (J9WallClock::start(), + // j9/j9WallClock.cpp): each +// wake (adaptive cadence, or earlier via onGCFinish()'s pthread_kill below) +// checks shouldRunPass() and calls runPass() if it says so. The pause-time +// pacing controller sleeps for _effective_cadence_ns rather than the fixed +// PASS_CADENCE_NS, so a +// controller-driven relaxed cadence (updatePacing()) actually shortens how +// long an idle, no-GC-event search waits between passes, not just +// shouldRunPass()'s own comparison. void ReferenceChainTracker::threadLoop() { struct Cleanup { ReferenceChainTracker *tracker; ~Cleanup() { - // No cached-class cleanup needed before detaching: _cached_object_class is a global ref, - // deliberately valid across attach/detach cycles (see its own comment in referenceChains.h) - - // unlike the per-attach local ref it replaced, which this destructor used to have to clear - // here. + // No cached-class cleanup needed before detaching: + // _cached_object_class is a global ref, deliberately valid across + // attach/detach cycles (see its own comment in referenceChains.h) - + // unlike the per-attach local ref it replaced, which this destructor + // used to have to clear here. VM::detachThread(); } } cleanup{this}; JNIEnv *jni = VM::attachThread("java-profiler ReferenceChains"); jvmtiEnv *jvmti = VM::jvmti(); if (jni == nullptr) { - // AttachCurrentThreadAsDaemon() failed - mirror pollWatchedTargets()'s own jni==nullptr early - // return rather than letting a null JNIEnv flow into - // runPass()/resolveLoadedClasses()/expandFrontier()/ releaseSearchTags() below: those only - // guard their DeleteLocalRef() calls on `jni != nullptr`, so without this check every - // GetLoadedClasses()/GetObjectsWithTags() local ref returned on this (permanently un-attached) - // thread would leak for the rest of the process's lifetime. + // AttachCurrentThreadAsDaemon() failed - mirror pollWatchedTargets()'s + // own jni==nullptr early return rather than letting a null JNIEnv flow + // into runPass()/resolveLoadedClasses()/expandFrontier()/ + // releaseSearchTags() below: those only guard their DeleteLocalRef() + // calls on `jni != nullptr`, so without this check every + // GetLoadedClasses()/GetObjectsWithTags() local ref returned on this + // (permanently un-attached) thread would leak for the rest of the + // process's lifetime. Nothing this thread does is safe without a live + // JNIEnv, so give up on the whole loop rather than retrying per + // iteration - detachThread() in Cleanup is a safe no-op if attach never + // actually succeeded. Log::warn("ReferenceChains: VM::attachThread failed; BFS thread exiting"); + // Clear the flag startThread() published before pthread_create (see + // its startup-race comment): without this, _running stays true forever + // with no live thread, and every later startThread() early-returns on + // the flag - the tracker never runs again for the process lifetime. + // Resetting here lets the next Profiler::start() retry. + _running.store(false, std::memory_order_release); return; } DEBUG_ONLY(rcDebugLevelRefresh(true)); // apply the override file before the first log line @@ -450,8 +1005,18 @@ void ReferenceChainTracker::threadLoop() { int iteration = 0; while (_running.load(std::memory_order_acquire)) { - // Fixed ~1s cadence, no early wake on GC (see onGCFinish()'s own comment) - stopThread() still - // interrupts this via its own pthread_kill so shutdown stays prompt. + // Fixed ~1s cadence, no early wake on GC (see onGCFinish()'s own + // comment) - stopThread() still interrupts this via its own + // pthread_kill so shutdown stays prompt. + // Urgency-driven dynamic tuning: as secondsToOOM() falls within + // OOM_RAMP_START_S of projected exhaustion, ramp the per-pass pause + // target and cadence exponentially toward their ceilings (see + // OOM_RAMP_START_S/URGENT_PAUSE_TARGET_MS/URGENT_CADENCE_NS's own + // comments) - slow at the 30-minute mark, aggressive right before OOM. + // secondsToOOM() itself already gates on a confirmed rising trend (its + // NOT_RISING check), so a non-negative value here is real growth, not + // noise. The PID controller is reconstructed whenever the (rounded) + // target changes so its ceiling tracks the new value. double seconds_to_oom = LivenessTracker::instance()->secondsToOOM(); bool urgent = seconds_to_oom >= 0 && seconds_to_oom < OOM_RAMP_START_S; long target_ms = _pause_target_ms; @@ -460,13 +1025,18 @@ void ReferenceChainTracker::threadLoop() { double x = 1.0 - seconds_to_oom / OOM_RAMP_START_S; // 0 at 30min out, 1 at OOM target_ms = std::lround(_pause_target_ms * std::pow((double)URGENT_PAUSE_TARGET_MS / std::max(_pause_target_ms, 1L), x)); - // Ramp from the fixed configured cadence, not the currently-adaptive _effective_cadence_ns - - // using the live value as the ramp's own moving anchor would compound the exponent across - // iterations instead of tracking urgency directly from a stable baseline. + // Ramp from the fixed configured cadence, not the currently-adaptive + // _effective_cadence_ns - using the live value as the ramp's own + // moving anchor would compound the exponent across iterations instead + // of tracking urgency directly from a stable baseline. cadence_ns = (u64)std::llround((double)PASS_CADENCE_NS * std::pow((double)URGENT_CADENCE_NS / (double)PASS_CADENCE_NS, x)); - // While urgent, the ramp owns _effective_cadence_ns outright so shouldRunPass()'s cadence - // gate and the per-pass log actually reflect it. + // While urgent, the ramp owns _effective_cadence_ns outright so + // shouldRunPass()'s cadence gate and the per-pass log actually + // reflect it. updatePacing()'s own overflow-driven widen/narrow + // adjustment (see _effective_cadence_ns's header comment) resumes + // sole ownership the instant urgency clears - this block simply stops + // touching the field then, so there is nothing to snap back from. _effective_cadence_ns = cadence_ns; } if (target_ms != _effective_pause_target_ms) { @@ -474,10 +1044,15 @@ void ReferenceChainTracker::threadLoop() { _pause_pid = PidController((u64)std::max(_effective_pause_target_ms, 0L), 10, 1, 2, 1, 5.0); } - // Once in the ramp window, hold the budget ceiling raised for the urgency episode's entire - // duration rather than only right before OOM: the process is likely to die anyway, so it's - // worth spending whatever budget it takes to collect good diagnostic data for as long as we - // have. + // Once in the ramp window, hold the budget ceiling raised for the + // urgency episode's entire duration rather than only right before OOM: + // the process is likely to die anyway, so it's worth spending whatever + // budget it takes to collect good diagnostic data for as long as we + // have. The boost is applied ONCE when urgency begins - the ramp's + // rounded pause target drifts every tick, and multiplying per change + // would reach the cap in two ticks and mask the configured budget - and + // the configured budget is restored the moment urgency clears, so a + // post-episode search cannot keep running with the inflated ceiling. if (urgent && !_urgency_budget_boosted) { _urgency_budget_boosted = true; _budget = std::min(_budget * 4, MAX_REFERENCE_CHAINS_BUDGET); @@ -491,33 +1066,52 @@ void ReferenceChainTracker::threadLoop() { "budget=%d", _budget); } // Third trigger for LivenessTracker::cleanup_table() (see - // LivenessTracker::maybeForceCleanup()'s own comment): track()'s table-overflow branch and - // flush_table()'s JFR cadence can both starve under ObjectSampler's PID-controlled sampling - // interval, leaving hasLeakSignal() below stuck on a stale population history no matter how - // long a real leak keeps growing. + // LivenessTracker::maybeForceCleanup()'s own comment): track()'s + // table-overflow branch and flush_table()'s JFR cadence can both starve + // under ObjectSampler's PID-controlled sampling interval, leaving + // hasLeakSignal() below stuck on a stale population history no matter + // how long a real leak keeps growing. This thread already wakes every + // ~1s with a live JNIEnv, so it doubles as that fallback tick - cheap, + // and a no-op unless 30s have actually elapsed with a GC in between (see + // that method for the exact gate). u64 wake_now_ns = OS::nanotime(); LivenessTracker::instance()->maybeForceCleanup(wake_now_ns); - // No fast-path skip here: shouldRunPass() below already returns false cheaply (a couple of - // atomic loads/comparisons, no JVMTI call) for a RUNNING search with no new GC and cadence not - // yet elapsed. + // No fast-path skip here: shouldRunPass() below already returns false + // cheaply (a couple of atomic loads/comparisons, no JVMTI call) for a + // RUNNING search with no new GC and cadence not yet elapsed. An earlier + // revision additionally gated this on hasLeakSignal() (LivenessTracker's + // population-trend signal, also used by canAffordNewSearch() below to gate + // the first-ever search and every restart), but that signal answers "is + // there a leak candidate right now", which is unrelated to whether an + // already-RUNNING search's own frontier still has pending work - gating a + // RUNNING search's every pass on it would stall that search's own + // convergence for as long as no leak candidate happens to be visible, + // even with GC epochs advancing or cadence elapsed. hasLeakSignal() + // remains the right gate for starting a *new* search, whether that is the + // first one ever or a restart of a *terminal* one (shouldRunPass()'s own + // canAffordNewSearch() call). u64 now_ns = OS::nanotime(); - // Re-check the runtime debug-level override file (~1s TTL; see rcDebugLevel.h) - never in heap - // callbacks, which only read the cached atomic. + // Re-check the runtime debug-level override file (~1s TTL; see + // rcDebugLevel.h) - never in heap callbacks, which only read the + // cached atomic. DEBUG_ONLY(rcDebugLevelRefresh()); - // Hand this iteration's ramp state to shouldRunPass() before it decides - the canary-backoff - // gate is bypassed while the OOM urgency ramp is active (see _oom_ramp_active's own comment) - - // and raise LivenessTracker's tracking admission to 100% for the same ramp (see - // setUrgentTracking()'s own comment, livenessTracker.h): same state, same iteration, so the - // boost tracks the ramp exactly, engaging and releasing together. + // Hand this iteration's ramp state to shouldRunPass() before it decides + // - the canary-backoff gate is bypassed while the OOM urgency ramp is + // active (see _oom_ramp_active's own comment) - and raise + // LivenessTracker's tracking admission to 100% for the same ramp + // (see setUrgentTracking()'s own comment, livenessTracker.h): same state, + // same iteration, so the boost tracks the ramp exactly, engaging and + // releasing together. _oom_ramp_active = urgent; LivenessTracker::instance()->setUrgentTracking(urgent); bool should_run = shouldRunPass(now_ns); - // Only sleep when idle (no pass will run). When a canary search is active or a pass is about to - // run, skip the sleep to run passes back-to-back. + // Only sleep when idle (no pass will run). When a canary + // search is active or a pass is about to run, skip the + // sleep to run passes back-to-back. if (!should_run && cadence_ns > 0) { OS::sleep(cadence_ns); if (!_running.load(std::memory_order_acquire)) { @@ -525,8 +1119,9 @@ void ReferenceChainTracker::threadLoop() { } now_ns = OS::nanotime(); } - // Log the loop state only when a pass is actually going to run - the idle wakes (should_run == - // false) are the common steady state and logging them every second is pure noise. + // Log the loop state only when a pass is actually going to run - the idle + // wakes (should_run == false) are the common steady state and logging them + // every second is pure noise. if (should_run) { TEST_LOG_SUMMARY("ReferenceChainTracker::threadLoop iteration=%d shouldRunPass=%d searchState=%d " "passesRun=%d effectiveCadenceNs=%llu effectiveBudget=%d gcFinishEpoch=%llu " @@ -537,8 +1132,12 @@ void ReferenceChainTracker::threadLoop() { (unsigned long long)(now_ns - _last_pass_ns)); runPassSerialized(jvmti, jni); } - // Target-selection bridging step: poll once per scheduling cycle, after runPass() - so this - // poll always sees the most recent pass's tagging (see pollWatchedTargets()'s own comment). + // Target-selection bridging step: poll once per scheduling cycle, after + // runPass() - so this poll always sees the most recent pass's tagging (see + // pollWatchedTargets()'s own comment). Unconditional, not gated on + // shouldRunPass()'s decision above: a candidate discovered by an + // earlier pass may still be waiting for its first poll even on a cycle + // where this cycle's own pass was skipped. pollWatchedTargetsSerialized(jvmti, jni); } } @@ -555,33 +1154,50 @@ void ReferenceChainTracker::onGCStart() { if (!_enabled) { return; } - // JVMTI spec: only Memory Management category calls (Allocate/Deallocate) are allowed from inside - // this callback - nothing else may run here. + // JVMTI spec: only Memory Management category calls (Allocate/Deallocate) + // are allowed from inside this callback - nothing else may run here. GCCallbackGuard guard; atomicIncRelaxed(_gc_start_epoch, (u64)1); } void ReferenceChainTracker::onGCFinish() { if (!_enabled) { + // Post-stop cost of the vmEntry trampoline is this single check per GC: + // stop() deliberately leaves the JVMTI GC notifications enabled (see + // ReferenceChainTracker::stop()) so a subsequent start() does not need + // to re-derive the event-enable state, and the tracker's lazy-start + // contract means this guard is the only per-GC work while disabled. return; } GCCallbackGuard guard; - // Heap-category JVMTI calls are forbidden from GC callbacks. + // Design doc's Triggering section: GC callbacks are only a scheduling + // *signal*, never a pass's execution vehicle (Heap-category JVMTI calls + // are forbidden here - see this file's header comment). Deliberately just + // bookkeeping - no pthread_kill/early wake here. threadLoop() below wakes + // on its own fixed ~1s cadence and reads this epoch then; waking it early + // on every GC gains at most ~1s of latency but, under any GC-heavy + // workload, collapses the loop's cadence to GC frequency instead (each + // early wake is itself a full iteration's worth of shouldRunPass()/ + // pollWatchedTargets() work), which is not worth the latency win. atomicIncRelaxed(_gc_finish_epoch, (u64)1); } bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { if (!_search_started) { - // Same gate as a restart (canAffordNewSearch() below) - a brand-new tracker must not pay for - // the first whole-heap walk/tagging pass either when there is no leak candidate to justify it. + // Same gate as a restart (canAffordNewSearch() below) - a brand-new + // tracker must not pay for the first whole-heap walk/tagging pass either + // when there is no leak candidate to justify it. The pain-budget half is + // always a no-op here (nothing has ever been spent yet), so this reduces + // to hasLeakSignal() in practice, but sharing the one gate keeps both + // call sites from drifting apart. bool afford = canAffordNewSearch(now_ns); TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass search_not_started " "canAffordNewSearch=%d", (int)afford); if (!afford) { return false; } - // This episode's one urgency-authorized search (_urgent_search_spent's own comment, - // referenceChains.h) is the one about to start. + // This episode's one urgency-authorized search (_urgent_search_spent's + // own comment, referenceChains.h) is the one about to start. _urgent_search_spent = _urgent_latched; TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (search not started yet)"); return true; // nothing has run yet - always worth taking the first pass @@ -589,25 +1205,29 @@ bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { if (_search_state != SearchState::RUNNING) { // Terminal outcome already reached (runPass()'s Termination section). if (!_tags_released) { - // releaseSearchTags() failed to confirm every live tag this search owned was actually cleared - // - restartSearch() must never run until that is confirmed (see _tags_released's own - // comment), so return true unconditionally here: that drives threadLoop() to call runPass() - // again, whose terminal-state branch retries the release, rather than letting - // canAffordNewSearch()/restartSearch() below run ahead of it. + // releaseSearchTags() failed to confirm every live tag this search + // owned was actually cleared - restartSearch() must never run until + // that is confirmed (see _tags_released's own comment), so return + // true unconditionally here: that drives threadLoop() to call + // runPass() again, whose terminal-state branch retries the release, + // rather than letting canAffordNewSearch()/restartSearch() below run + // ahead of it. TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (retrying tag " "release before restart is allowed)"); return true; } - // Charge the finished search's accumulated safepoint cost BEFORE the restart gate - - // canAffordNewSearch() must see the cost of the search that just ended, otherwise an expensive - // search earns one free immediate successor (the accumulator is spent here, once per search; + // Charge the finished search's accumulated safepoint cost BEFORE the + // restart gate - canAffordNewSearch() must see the cost of the search + // that just ended, otherwise an expensive search earns one free + // immediate successor (the accumulator is spent here, once per search; // repeated terminal visits spend a zeroed accumulator). _safepoint_pain_budget.spend(_search_pain_ms); _search_pain_ms = 0; - // Restart (this class's own header comment) if the pain budget has drained and there is still - // (or again) a leak indication to chase - canAffordNewSearch() is always true when - // LivenessTracker's population trends are not in use at all, so this only ever changes behavior - // for a search that already ran once. + // Restart (this class's own header comment) if the pain budget has + // drained and there is still (or again) a leak indication to chase - + // canAffordNewSearch() is always true when LivenessTracker's population + // trends are not in use at all, so this only ever changes behavior for a + // search that already ran once. if (canAffordNewSearch(now_ns)) { // Same entitlement bookkeeping as the first-search branch above. _urgent_search_spent = _urgent_latched; @@ -615,8 +1235,9 @@ bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (restarting search)"); return true; } - // No log here: a terminal search waiting for a restart to become warranted is the common idle - // state, re-evaluated every second, so logging it is pure per-second noise (see threadLoop()). + // No log here: a terminal search waiting for a restart to become + // warranted is the common idle state, re-evaluated every second, so + // logging it is pure per-second noise (see threadLoop()). TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass terminal_blocked " "tags_released=%d safepoint_pain=%d search_state=%d", (int)_tags_released, @@ -624,15 +1245,17 @@ bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { (int)_search_state); return false; } - // Canary search active with candidates still to find - computed ahead of the pain-budget check - // below so the refill-rate raise and the backoff gate further down agree on the same snapshot of - // _candidate_found_bits. + // Canary search active with candidates still to find - computed ahead of + // the pain-budget check below so the refill-rate raise and the backoff + // gate further down agree on the same snapshot of _candidate_found_bits. bool canary_active = _candidate_count > 0 && __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count; - // Adaptive CPU budget: 100x refill while a canary chase is open - NOT a rate control (the canary - // lane's rate is bounded by _canary_backoff_ns's progress-driven exponential backoff, see its own - // comment) but a double-throttle guard: the base refill rate is tuned for the ordinary ~1 pass/s - // whole-graph cadence and would otherwise starve a chase the backoff has already paced. + // Adaptive CPU budget: 100x refill while a canary chase is open - NOT a + // rate control (the canary lane's rate is bounded by _canary_backoff_ns's + // progress-driven exponential backoff, see its own comment) but a + // double-throttle guard: the base refill rate is tuned for the ordinary + // ~1 pass/s whole-graph cadence and would otherwise starve a chase the + // backoff has already paced. 1x in every other mode. double multiplier = canary_active ? CANARY_PAIN_BUDGET_REFILL_MULTIPLIER : 1.0; _cpu_pain_budget.setRefillRate( @@ -648,43 +1271,56 @@ bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { return false; } if (canary_active) { - // Canary-lane pacing: the chase's rate bound - work-scaled spacing (_canary_backoff_mult's own - // comment for the law and the live burn it bounds). + // Canary-lane pacing: the chase's rate bound - work-scaled spacing + // (_canary_backoff_mult's own comment for the law and the live burn it + // bounds). Deliberately placed ABOVE the gc-finish-epoch trigger below: + // a GC-heavy workload bumps the epoch on virtually every wake (minor + // young GCs included), so letting the epoch trigger bypass the backoff + // would make the backoff unreachable exactly on the GC-churning + // deployments that burn the most. The pass that eventually runs sees + // whatever the graph looks like then - freshness is not lost, only + // re-checked at the paced rate. The OOM urgency ramp is the one + // override (see _oom_ramp_active's own comment). u64 spacing_ns = - (u64)_canary_backoff_mult * _canary_pass_ema_ms * 1000000ULL; - // mult == 1 (fresh chase, or last pass made progress) means the gate is OFF - the chase runs at - // its natural pass rate, one pass starting as soon as the last ended. + (u64)_canary_backoff_mult * _canary_pass_ema_ns; + // mult == 1 (fresh chase, or last pass made progress) means the gate is + // OFF - the chase runs at its natural pass rate, one pass starting as + // soon as the last ended. if (!_oom_ramp_active && _canary_backoff_mult > 1 && now_ns - _last_canary_pass_ns < spacing_ns) { TEST_LOG("ReferenceChainTracker::shouldRunPass held off by canary " - "backoff mult=%d ema_ms=%llu since_last_pass=%llums", + "backoff mult=%d ema_ns=%llu since_last_pass=%llums", _canary_backoff_mult, - (unsigned long long)_canary_pass_ema_ms, + (unsigned long long)_canary_pass_ema_ns, (unsigned long long)((now_ns - _last_canary_pass_ns) / 1000000ULL)); return false; } TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (canary search, " - "%d/%d candidates found, backoff_mult=%d ema_ms=%llu)", + "%d/%d candidates found, backoff_mult=%d ema_ns=%llu)", (int)__builtin_popcountll(_candidate_found_bits), (int)_candidate_count, _canary_backoff_mult, - (unsigned long long)_canary_pass_ema_ms); + (unsigned long long)_canary_pass_ema_ns); return true; } u64 gc_finish_epoch = gcFinishEpoch(); if (gc_finish_epoch != _last_pass_gc_finish_epoch) { - // Triggering section: "a GC just happened, a pass may be worth running soon". + // Triggering section: "a GC just happened, a pass may be worth running + // soon". TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (gcFinishEpoch=%llu != " "lastPassGcFinishEpoch=%llu)", (unsigned long long)gc_finish_epoch, (unsigned long long)_last_pass_gc_finish_epoch); return true; } - // Pause-time pacing controller: compares against _effective_cadence_ns, not the fixed - // PASS_CADENCE_NS - see that field's own comment (referenceChains.h) for how updatePacing() - // widens or relaxes it from the measured pause-time signal. + // Pause-time pacing controller: compares against _effective_cadence_ns, not + // the fixed PASS_CADENCE_NS - see that + // field's own comment (referenceChains.h) for how updatePacing() widens or + // relaxes it from the measured pause-time signal. bool cadence_elapsed = now_ns - _last_pass_ns >= _effective_cadence_ns; - // Only log when the cadence actually elapsed (a pass will run). + // Only log when the cadence actually elapsed (a pass will run). The + // not-yet-elapsed case is the common idle wake and logging it every second + // is noise. if (cadence_elapsed) { TEST_LOG_SUMMARY("ReferenceChainTracker::shouldRunPass -> true (now_ns=%llu last_pass_ns=%llu " "delta=%llu effectiveCadenceNs=%llu)", @@ -695,10 +1331,14 @@ bool ReferenceChainTracker::shouldRunPass(u64 now_ns) { return cadence_elapsed; } -// Search restart gate (this class's own header comment). Deliberately a probe (max=1) rather than -// reusing pollWatchedTargets()'s own selectLeakCandidates() call - that one runs after runPass() in -// threadLoop()'s own iteration and needs the *list* to poll each candidate's tag; this only needs -// to know whether at least one exists. +// Search restart gate (this class's own header comment). Deliberately a +// probe (max=1) rather than reusing pollWatchedTargets()'s own +// selectLeakCandidates() call - that one runs after runPass() in +// threadLoop()'s own iteration and needs the *list* to poll each candidate's +// tag; this only needs to know whether at least one exists. +// Latching, hysteretic read of LivenessTracker::secondsToOOM() - see +// _urgent_latched's own comment (referenceChains.h) for why a bare threshold +// comparison here flaps, and OOM_URGENT_RELEASE_S for the release bar. bool ReferenceChainTracker::isUrgent() const { double seconds_to_oom = LivenessTracker::instance()->secondsToOOM(); if (seconds_to_oom >= 0 && seconds_to_oom < OOM_URGENT_THRESHOLD_S) { @@ -714,8 +1354,9 @@ bool ReferenceChainTracker::isUrgent() const { return true; } if (_urgent_latched) { - // Negative means "no rising trend to project from" (secondsToOOM()'s own unknown/NOT_RISING - // encoding), which counts toward release just like a comfortably distant projection does. + // Negative means "no rising trend to project from" (secondsToOOM()'s own + // unknown/NOT_RISING encoding), which counts toward release just like a + // comfortably distant projection does. if (seconds_to_oom < 0 || seconds_to_oom >= OOM_URGENT_RELEASE_S) { if (++_urgent_release_ticks >= URGENT_RELEASE_CONSECUTIVE) { _urgent_latched = false; @@ -729,8 +1370,8 @@ bool ReferenceChainTracker::isUrgent() const { return false; } } else { - // Between the two bars, or a single noisy reading past the release bar followed by one that - // is not - neither releases the latch. + // Between the two bars, or a single noisy reading past the release bar + // followed by one that is not - neither releases the latch. _urgent_release_ticks = 0; } return true; @@ -740,20 +1381,27 @@ bool ReferenceChainTracker::isUrgent() const { bool ReferenceChainTracker::hasLeakSignal() { if (!LivenessTracker::instance()->gcGenerationsEnabled()) { - // No population-trend signal to gate on at all - see this method's own header comment for why - // that means "always true" here. + // No population-trend signal to gate on at all - see this method's own + // header comment for why that means "always true" here. return true; } double seconds_to_oom = LivenessTracker::instance()->secondsToOOM(); - // isUrgent() is called unconditionally, not short-circuited behind _urgent_search_spent: it is - // what maintains the latch/release counter, so skipping it would freeze the episode state (see - // _urgent_latched). + // isUrgent() is called unconditionally, not short-circuited behind + // _urgent_search_spent: it is what maintains the latch/release counter, so + // skipping it would freeze the episode state (see _urgent_latched). bool urgent = isUrgent(); if (urgent && !_urgent_search_spent) { - // Heap-wide floor is rising fast enough that OOM is imminent - don't wait for a specific klass - // to clear selectLeakCandidates()'s own per-klass ring-fill/hysteresis gate; see - // OOM_URGENT_THRESHOLD_S's own comment (referenceChains.h) for why that gate alone is too slow - // here. + // Heap-wide floor is rising fast enough that OOM is imminent - don't + // wait for a specific klass to clear selectLeakCandidates()'s own + // per-klass ring-fill/hysteresis gate; see OOM_URGENT_THRESHOLD_S's own + // comment (referenceChains.h) for why that gate alone is too slow here. + // canAffordNewSearch() can still defer this via the pain-budget check it + // runs before calling this method (this method's own header comment). + // + // Only until this episode's one search has been started + // (_urgent_search_spent): past that point the per-klass probe below is + // the sole remaining trigger, so a completed urgent search is not torn + // down and restarted from scratch on the very next tick. TEST_LOG_SUMMARY("ReferenceChainTracker::hasLeakSignal -> true (urgent, " "secondsToOOM=%.1f)", seconds_to_oom); @@ -779,34 +1427,40 @@ bool ReferenceChainTracker::canAffordNewSearch(u64 now_ns) { return hasLeakSignal(); } -// Search restart (this class's own header comment). Called only from shouldRunPass() once -// canAffordNewSearch() has approved it, immediately before returning true for this same iteration - -// runPass() then sees _search_started == false and takes the first-pass branch, exactly like a +// Search restart (this class's own header comment). Called only from +// shouldRunPass() once canAffordNewSearch() has approved it, immediately +// before returning true for this same iteration - runPass() then sees +// _search_started == false and takes the first-pass branch, exactly like a // brand-new tracker. void ReferenceChainTracker::restartSearch() { - // Only called once shouldRunPass() has confirmed _tags_released - never while a prior search's - // release might still be pending (see _tags_released's own comment): resetting _next_tag to 1 / - // the frontier table below while some object could still hold this search's now- ambiguous tag - // would let the restarted search's fresh tags collide with it. + // Only called once shouldRunPass() has confirmed _tags_released - never + // while a prior search's release might still be pending (see + // _tags_released's own comment): resetting _next_tag to 1 / the frontier + // table below while some object could still hold this search's now- + // ambiguous tag would let the restarted search's fresh tags collide with + // it. assert(_tags_released && "restartSearch() must not run before releaseSearchTags() has " "confirmed every live tag was cleared"); - // The finishing search's accumulated safepoint cost is spent by the terminal restart gate in - // shouldRunPass() BEFORE it calls this (see the gate's comment: the gate must see the finished - // search's cost), so no spend happens here. + // The finishing search's accumulated safepoint cost is spent by the + // terminal restart gate in shouldRunPass() BEFORE it calls this (see the + // gate's comment: the gate must see the finished search's cost), so no + // spend happens here. if (_frontier != nullptr) { _frontier->resetForRestart(); } _next_tag = 1; - // Hop-edge label cache: keyed by raw class tags, which survive a restart (the shared class-tag - // allocator is deliberately not reset - see this method's own declaration comment) - but the - // frontier entries referencing them do not, and a restart is the natural bounded clear point for - // a cache capped by HOP_LABEL_CLASS_CACHE_CAP wholesale. + // Hop-edge label cache: keyed by raw class tags, which survive a restart + // (the shared class-tag allocator is deliberately not reset - see this + // method's own declaration comment) - but the frontier entries referencing + // them do not, and a restart is the natural bounded clear point for a + // cache capped by HOP_LABEL_CLASS_CACHE_CAP wholesale. _hop_label_cache.clear(); - // The shared class-tag counter (classTagAllocator.h)/_class_tags intentionally untouched - see - // this method's own declaration comment (referenceChains.h). + // The shared class-tag counter (classTagAllocator.h)/_class_tags + // intentionally untouched - see this method's own declaration comment + // (referenceChains.h). _search_started = false; store(_search_state, (u8)SearchState::RUNNING); @@ -823,16 +1477,29 @@ void ReferenceChainTracker::restartSearch() { _static_anchor_index_tags.clear(); _anchor_container_cursor = 0; _anchor_other_cursor = 0; - // Fresh lane: nothing admits before the search does, so nothing can have a pending first look - // either. + // Fresh lane: nothing admits before the search does, so nothing can + // have a pending first look either. _static_anchor_fresh_queue.clear(); - // Discovered-instance tags are FRONTIER tags - the reset above just invalidated every one of them - // (fresh tags restart from 1). + // Discovered-instance tags are FRONTIER tags - the reset above just + // invalidated every one of them (fresh tags restart from 1). Leaving + // the slots populated lets pollWatchedTargets() resolve stale tags into + // whatever unrelated object the new search assigns them to: a dead + // slot fails reconstructChain() (observed on-pod round 13: 'buildChainEvent + // failed ... reconstructChain failed for target_tag=8851'), and a live + // one emits a chain event for the WRONG OBJECT (the likely origin of + // the earlier session's noise-[B event). _class_shape_cache is + // deliberately NOT cleared here: class tags are stable for the JVM's + // lifetime (the class-tag allocator is not reset), so a classification + // remains valid across searches. memset(_candidate_discovered_tags, 0, sizeof(_candidate_discovered_tags)); memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); - // Both keyed by frontier tags this restart is about to invalidate (fresh tags start again from 1) - // - a stale entry surviving past a restart would be compared against whatever unrelated object - // the new search has since reassigned that tag to. + // Both keyed by frontier tags this restart is about to invalidate (fresh + // tags start again from 1) - a stale entry surviving past a restart would + // be compared against whatever unrelated object the new search has since + // reassigned that tag to. _watched_leak_klass_ids itself is left alone: + // it reflects LivenessTracker's own growth signal, unrelated to this + // search's lifecycle, and simply gets refreshed again on the next + // pollWatchedTargets() tick regardless. _leak_signature_totals.clear(); _leak_signature_prev_totals.clear(); _leak_parent_fanout.clear(); @@ -841,30 +1508,38 @@ void ReferenceChainTracker::restartSearch() { _last_pass_gc_finish_epoch = 0; store(_last_pass_ns, (u64)0); store(_passes_run, 0); - // Reset back to their just-constructed values (0 / -1) like every other per-search field this - // method touches: resolveLoadedClasses() and admitStaticFieldRoots() must both run - // unconditionally on the restarted search's first pass, exactly as they do for a brand-new - // tracker. + // Reset back to their just-constructed values (0 / -1) like every other + // per-search field this method touches: resolveLoadedClasses() and + // admitStaticFieldRoots() must both run unconditionally on the restarted + // search's first pass, exactly as they do for a brand-new tracker. _last_resolved_class_count = 0; _last_static_field_class_count = -1; - // _resolved_chains is intentionally left intact: a chain resolved by the finishing search stays - // cached (and keeps being re-emitted on every dump) across the restart, since it describes a - // sample that is still live. + // _resolved_chains is intentionally left intact: a chain resolved by the + // finishing search stays cached (and keeps being re-emitted on every dump) + // across the restart, since it describes a sample that is still live. The + // restarted search re-tags that sample under a fresh _search_start_ns, and + // pollWatchedTargets() refreshes the cached entry then (its own comment); + // it prunes the entry if the sample has since been collected. } void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, JNIEnv *jni) { - // Every field touched below is otherwise only ever mutated by the BFS thread itself - // (threadLoop()/runPass()/pollWatchedTargets()) - without stopping it first, a pass already in - // flight on that thread can observe this reset only partially, or overwrite it right back (e.g. - // finish a pass that was already headed for SearchState::ABANDONED after this method has just - // forced SearchState::RUNNING below), a race found in practice, not just in theory. + // Every field touched below is otherwise only ever mutated by the BFS + // thread itself (threadLoop()/runPass()/pollWatchedTargets()) - without + // stopping it first, a pass already in flight on that thread can observe + // this reset only partially, or overwrite it right back (e.g. finish a + // pass that was already headed for SearchState::ABANDONED after this + // method has just forced SearchState::RUNNING below), a race found in + // practice, not just in theory. stopThread() (now that it can abort an + // in-flight JVMTI walk promptly - see its own comment) makes this a cheap, + // clean stop/reset/restart rather than an indefinite wait. stopThread(); - // Clear every live tag this search still holds before resetting - the same ordering - // restartSearch() itself requires (its own assert), so a stale tag from whatever search a - // previous test left running cannot collide with the fresh search's own tags once _next_tag is - // rewound below. + // Clear every live tag this search still holds before resetting - the + // same ordering restartSearch() itself requires (its own assert), so a + // stale tag from whatever search a previous test left running cannot + // collide with the fresh search's own tags once _next_tag is rewound + // below. if (jvmti != nullptr && jni != nullptr) { releaseSearchTags(jvmti, jni); } @@ -872,22 +1547,26 @@ void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, _safepoint_pain_budget.spend(_search_pain_ms); _search_pain_ms = 0; - // Reset the pain budget entirely so a fresh test starts from zero debt, independent of how much - // wall-clock time has elapsed since the last test's spend(). + // Reset the pain budget entirely so a fresh test starts from zero debt, + // independent of how much wall-clock time has elapsed since the last + // test's spend(). Without this, a fast CI runner (musl, small heap, + // no GC pauses) may not have drained enough debt between tests. _safepoint_pain_budget = PainBudget(_pain_budget_refill_rate); - // Mirror the reset for the non-safepoint budget - same test-isolation rationale as - // _safepoint_pain_budget above. + // Mirror the reset for the non-safepoint budget - same test-isolation + // rationale as _safepoint_pain_budget above. _cpu_pain_budget = PainBudget(_pain_budget_refill_rate); - // Same test-isolation rationale: a latched urgency episode left behind by an earlier test would - // otherwise deny this one its own urgency-authorized search (see _urgent_search_spent). + // Same test-isolation rationale: a latched urgency episode left behind by + // an earlier test would otherwise deny this one its own + // urgency-authorized search (see _urgent_search_spent). _urgent_latched = false; _urgent_release_ticks = 0; _urgent_search_spent = false; if (_frontier != nullptr) { - // Rebuilds the table at this test's own _configured_frontier_cap, undoing any smaller framecap= - // an earlier test left it permanently sized at (this class's own header comment on - // @TestMethodOrder) - restartSearch()'s production path only calls the cheaper + // Rebuilds the table at this test's own _configured_frontier_cap, + // undoing any smaller framecap= an earlier test left it permanently + // sized at (this class's own header comment on @TestMethodOrder) - + // restartSearch()'s production path only calls the cheaper // resetForRestart() since it never needs to change the cap mid-JVM. _frontier->resetCapacityForTest(_configured_frontier_cap); } @@ -909,13 +1588,15 @@ void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, _anchor_container_cursor = 0; _anchor_other_cursor = 0; _static_anchor_fresh_queue.clear(); - // Same stale-frontier-tag hygiene as restartSearch() (see its comment): discovered tags are - // frontier tags, invalid across the test reset just as across a restart. + // Same stale-frontier-tag hygiene as restartSearch() (see its comment): + // discovered tags are frontier tags, invalid across the test reset just + // as across a restart. memset(_candidate_discovered_tags, 0, sizeof(_candidate_discovered_tags)); memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); - // Test-only extra: production restartSearch() keeps _class_shape_cache (class tags are - // JVM-lifetime-stable there), but test scenarios script class-tag values directly and a later - // test can reuse an earlier one for a different mock class - clear the cache between tests. + // Test-only extra: production restartSearch() keeps _class_shape_cache + // (class tags are JVM-lifetime-stable there), but test scenarios script + // class-tag values directly and a later test can reuse an earlier one + // for a different mock class - clear the cache between tests. _class_shape_cache.clear(); // Same reset rationale as restartSearch()'s own comment. _leak_signature_totals.clear(); @@ -930,15 +1611,17 @@ void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, _passes_since_last_candidate_progress = 0; _last_candidate_progress_mark = 0; _canary_backoff_mult = 1; - _canary_pass_ema_ms = 0; + _canary_pass_ema_ns = 0; _last_canary_pass_ns = 0; _canary_stuck_restart_count = 0; - // Same "just-constructed values" contract resetForRestart() already documents for these two - // fields - without it, a prior test's fully-swept (or partially-swept) state survives in this - // process-wide singleton (ReferenceChainTracker::instance()) and can wrongly skip - // admitStaticFieldRoots() entirely on this test's first pass if its resolved class count happens - // to match whatever an earlier test last left behind (found via a real gtest-suite-order failure, - // not hypothetical). + // Same "just-constructed values" contract resetForRestart() already + // documents for these two fields - without it, a prior test's fully-swept + // (or partially-swept) state survives in this process-wide singleton + // (ReferenceChainTracker::instance()) and can wrongly skip + // admitStaticFieldRoots() entirely on this test's first pass if its + // resolved class count happens to match whatever an earlier test last + // left behind (found via a real gtest-suite-order failure, not + // hypothetical). _last_resolved_class_count = 0; _last_static_field_class_count = -1; _static_field_sweep_cursor = 0; @@ -948,8 +1631,10 @@ void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); memset(_candidate_qualifying_tid_count, 0, sizeof(_candidate_qualifying_tid_count)); - // _candidate_parent_tags/_candidate_referrer_klasses/_candidate_depths will be filled at pruning - // time. + // _candidate_parent_tags/_candidate_referrer_klasses/_candidate_depths + // will be filled at pruning time. + // across a production restart, a test reset starts from a blank cache so + // one test's resolved chains cannot leak into the next. _resolved_chains_lock.lock(); _resolved_chains.clear(); _resolved_chains_lock.unlock(); @@ -957,9 +1642,9 @@ void ReferenceChainTracker::resetSearchStateForTest(jvmtiEnv *jvmti, _pending_abandoned_events.clear(); _pending_abandoned_events_lock.unlock(); - // Restart the BFS thread against this freshly reset state - startThread() itself clears - // _abort_pass_requested, so the new thread's very first pass is not instantly aborted by the flag - // stopThread() just set above. + // Restart the BFS thread against this freshly reset state - startThread() + // itself clears _abort_pass_requested, so the new thread's very first + // pass is not instantly aborted by the flag stopThread() just set above. startThread(); } @@ -967,8 +1652,8 @@ long ReferenceChainTracker::pendingExpandPositionForTest(jlong tag) const { if (tag == 0) { return -2; } - // _priority_expand drains first (expandFrontier()'s own comment), so its entries are reported as - // coming before _pending_expand's. + // _priority_expand drains first (expandFrontier()'s own comment), so its + // entries are reported as coming before _pending_expand's. long pos = 0; for (jlong queued : _priority_expand) { if (queued == tag) { @@ -1029,13 +1714,15 @@ jlong ReferenceChainTracker::tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, (void *)_frontier, (void *)jvmti, (void *)jni, (void *)obj); return 0; } - // Resolves the klass_id via the same GetClassSignature + normalizeClassSignature + - // Profiler::lookupClass sequence every consumer in this subsystem uses - // (ObjectSampler::recordAllocation(), LivenessTracker::resolveKlassId(), resolveClassMap() above) - // - the id space is load-bearing here: pollWatchedTargets() matches frontier entries against leak - // candidates by klass_id, and the candidate ids come from that signature-notation space - // (Class.getName()'s dot form is a DIFFERENT StringDictionary key - see - // find-klass-id-notation-mismatch). + // Resolves the klass_id via the same GetClassSignature + + // normalizeClassSignature + Profiler::lookupClass sequence every + // consumer in this subsystem uses (ObjectSampler::recordAllocation(), + // LivenessTracker::resolveKlassId(), resolveClassMap() above) - the + // id space is load-bearing here: pollWatchedTargets() matches frontier + // entries against leak candidates by klass_id, and the candidate ids + // come from that signature-notation space (Class.getName()'s dot form + // is a DIFFERENT StringDictionary key - see + // find-klass-id-notation-mismatch). Test-only, off-hot-path. u32 klass_id = 0; jclass klass = jni->GetObjectClass(obj); char *class_name = nullptr; @@ -1055,12 +1742,14 @@ jlong ReferenceChainTracker::tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, } jni->DeleteLocalRef(klass); - // Tags `obj` and inserts it as a frontier root (parent_tag=0, depth=0), exactly the convention - // runPass()'s heap-root callback path already uses (referenceChains.cpp's - // heapReferenceCallback(), referrer_tag_ptr == nullptr branch) - this lets a test drive the real - // BFS/chain- reconstruction logic (runPass()/pollWatchedTargets()/buildChainEvent()) against a - // known, caller-chosen live object, decoupled from whether the real root-seeded walk or - // LivenessTracker's probabilistic sampler happens to reach/select it on its own. + // Tags `obj` and inserts it as a frontier root (parent_tag=0, depth=0), + // exactly the convention runPass()'s heap-root callback path already uses + // (referenceChains.cpp's heapReferenceCallback(), referrer_tag_ptr == + // nullptr branch) - this lets a test drive the real BFS/chain- + // reconstruction logic (runPass()/pollWatchedTargets()/buildChainEvent()) + // against a known, caller-chosen live object, decoupled from whether the + // real root-seeded walk or LivenessTracker's probabilistic sampler happens + // to reach/select it on its own. jlong tag = tagObject(jvmti, obj); if (tag == 0) { TEST_LOG_SUMMARY("ReferenceChainTracker::tagAsRootForTest refused: " @@ -1074,9 +1763,5402 @@ jlong ReferenceChainTracker::tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, clearTag(jvmti, obj); return 0; } - // Discovery recording for this seam's decoupling contract. + // Discovery recording for this seam's decoupling contract. The leak-tag + // redesign ("no marker tags - using leak tags now", pollWatchedTargets()) + // left a representative without an ObjectSampler-tracked instance with no + // discovery channel at all: recordDiscoveredInstance() only fires from the + // leak-tag interception branch of heapReferenceCallback(), and + // tagLeakInstances() can only tag instances the sampler tracked. This + // seam's purpose is precisely to be decoupled from the sampler, so it + // records the directly-tagged root itself; pollWatchedTargets() then + // reconstructs the chain from the real frontier via + // buildDiscoveredInstanceChains()/buildChainEvent(). No-op while no + // candidate slot watches this klass yet (the caller must have driven at + // least one pollReferenceChainTargets0() first - see + // ReferenceChainTestSeamsTest's ordering comment). if (_candidate_count > 0) { recordDiscoveredInstance(klass_id, tag, false); } return tag; } + +// --------------------------------------------------------------------------- +// Heap-walk engine +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::resolveLoadedClasses(jvmtiEnv *jvmti, + JNIEnv *jni) { + // Profiler::start() resets the class-name StringDictionary + // (_class_map.clearAll(), profiler.cpp) whenever `reset || _start_time == + // 0` - which restarts its id namespace at 1, but does NOT touch any + // class's JVMTI-level class-object tag (JVM-level state, unrelated to our + // dictionary). Detect that reset via the dictionary's own generation + // counter and drop every id this table cached from the now-gone + // generation before the scan below - see _last_class_map_generation's own + // comment (referenceChains.h) for why leaving them in place would keep + // resolving heap references to the wrong (or nonexistent) class name. + u64 current_generation = Profiler::instance()->classMap()->generation(); + bool class_map_reset = current_generation != _last_class_map_generation; + if (class_map_reset) { + TEST_LOG_SUMMARY("ReferenceChainTracker::resolveLoadedClasses class_map generation " + "changed: old=%llu new=%llu - clearing _class_tags and " + "candidate klass_ids may be stale", + (unsigned long long)_last_class_map_generation, + (unsigned long long)current_generation); + _class_tags.clear(); + // Force the scan below to run even if GetLoadedClasses()'s count happens + // to match the last-seen count - -1 can never equal `class_count` + // (always >= 0), unlike 0 which is a legitimate "no classes loaded yet" + // starting value. + _last_resolved_class_count = -1; + _last_class_map_generation = current_generation; + } + + jclass *classes = nullptr; + jint class_count = 0; + if (jvmti->GetLoadedClasses(&class_count, &classes) != JVMTI_ERROR_NONE || + classes == nullptr) { + return; + } + + // Skip the per-class GetTag()/GetClassSignature() scan entirely once the + // loaded-class count has not CHANGED since the last time this ran it: + // every already-tagged class stays tagged forever (tags are never + // cleared once assigned - see _class_tags' own comment), so a resumed + // pass with no newly-loaded classes has nothing left to resolve. Without + // this, every single pass pays a full GetTag() call per loaded class + // (potentially thousands) even though almost all of them are already + // resolved, and that cost is invisible to the pause-time-SLO pacing + // controller (runPass()'s pass_wall_ticks measurement deliberately scopes + // out this call - see that field's own comment). + // + // Deliberately `!=`, not `>`: GetLoadedClasses()'s count is NOT monotonic + // - class unloading (a GC'd custom classloader, JSP/bytecode-macro + // recompilation, etc.) can shrink it. A `>` check would then stay + // permanently skipped once new classes are loaded back up to, but not + // past, a prior historical peak - e.g. 1000 classes loaded then unloaded + // down to 400, then 50 different new classes loaded (total 450, still + // below the 1000 peak) - silently leaving those 50 new classes' tag == 0 + // forever, so any object of theirs discovered by the BFS walk never + // resolves a referrer_klass. `!=` catches both directions; the only + // residual gap is the count-preserving unload-then-reload-same-count case, + // far narrower than the permanent gap `>` left open. + if (class_count != _last_resolved_class_count) { + for (jint i = 0; i < class_count; i++) { + jclass klass = classes[i]; + jlong tag = 0; + // Resolve if not yet tagged (ordinary case: a newly-loaded class), or + // unconditionally on a class-map reset (class_map_reset above) - a + // class already tagged from a prior generation still carries that same + // JVMTI tag (untouched by clearAll()), but the dictionary id it used to + // map to is gone, so its name must be re-resolved into the new + // generation too. + if (jvmti->GetTag(klass, &tag) == JVMTI_ERROR_NONE && + (tag == 0 || class_map_reset)) { + // Resolve its name now, via the same GetClassSignature + + // normalizeClassSignature + Profiler::lookupClass sequence + // ObjectSampler::recordAllocation() already uses + // (objectSampler.cpp:76-90), reused rather than re-derived. + char *class_name = nullptr; + if (jvmti->GetClassSignature(klass, &class_name, nullptr) == + JVMTI_ERROR_NONE && + class_name != nullptr) { + const char *name_slice = nullptr; + size_t name_len = 0; + if (ObjectSampler::normalizeClassSignature(class_name, &name_slice, + &name_len)) { + int id = Profiler::instance()->lookupClass(name_slice, name_len); + if (id != -1) { + TEST_LOG("ReferenceChainTracker::resolveClassMap id=%d name=%.*s", + id, (int)name_len, name_slice); + // Reuse the existing tag if this class was already tagged by a + // prior generation - only the resolved id needs refreshing, + // not the tag identity heapReferenceCallback() keys off of. + jlong class_tag = tag != 0 ? tag : nextClassTag(); + if (tag != 0 || + jvmti->SetTag(klass, class_tag) == JVMTI_ERROR_NONE) { + _class_tags.insert(class_tag, (u32)id); + } + } + } + jvmti->Deallocate((unsigned char *)class_name); + } + } + // GetLoadedClasses() hands back class_count fresh JNI local refs - + // delete each immediately rather than holding all of them alive at + // once, since class_count can run into the thousands. + if (jni != nullptr) { + jni->DeleteLocalRef(klass); + } + } + _last_resolved_class_count = class_count; + } else if (jni != nullptr) { + // Still owe DeleteLocalRef for every fresh local ref GetLoadedClasses() + // just handed back, even though the scan above was skipped. + for (jint i = 0; i < class_count; i++) { + jni->DeleteLocalRef(classes[i]); + } + } + jvmti->Deallocate((unsigned char *)classes); +} + +namespace { +// Per-runPass() state threaded through heapReferenceCallback() via +// FollowReferences' user_data parameter. Private to this .cpp - the type +// never needs to be visible in referenceChains.h since only runPass() +// constructs one and only heapReferenceCallback() reads it. +struct PassContext { + ReferenceChainTracker *tracker; + FrontierTable *frontier; + int hop_cap; + int budget; + int edges_admitted; + bool truncated; + + // Set only when `truncated` became true because frontier->insert() itself + // reported capacity exhaustion, as opposed to edges_admitted reaching + // budget. runPass() uses this to distinguish "this pass ran out + // of budget, more work remains for a later pass" (search stays RUNNING) + // from "the frontier table itself is full" (design doc's Termination + // section: grounds to ABANDON the whole search, not just this pass). + bool frontier_cap_hit; + + // ARRAY-HOLDER BATCHING: when non-null, expandFrontier() is driving a + // one-hop expansion of a batch of boundary objects passed to a single + // FollowReferences(initial_object=holder_array) call. heapReferenceCallback() + // then descends ONLY into objects whose tag is in this set (the boundary + // objects we deliberately put in the array), and returns "do not descend" + // (0) for everything else - so a freshly-admitted child is tagged but its + // own subtree is left for a later pass, and an already-expanded object from + // a prior pass is never re-traversed. Null on the whole-heap first pass + // (runPass()'s !_search_started branch) and IterateOverReachableObjects + // root enumeration, which keep the unconditional-descend behavior. + std::unordered_set *batch_tags = nullptr; + + // Rolling resume cursor for expandFrontier(): tracks the tag of the last + // batch entry that FollowReferences visited (the callback at the + // batch_tags descent-gate updates this). After FollowReferences returns + // truncated, expandFrontier() uses this to pop fully-processed entries + // from the source queue (mark EXPANDED) and leave only the + // partially-processed and unvisited entries for the next pass — same + // resumable-cursor pattern as admitStaticFieldRoots()'s sweep cursor. + // Without this, a truncated batch is retried in its entirety next pass: + // GetObjectsWithTags + FollowReferences re-walks already-expanded + // entries (their children are ALREADY_ADMITTED, so idempotent but + // wasteful — re-paying the full O(tag_map × batch) GOTW cost and the + // FollowReferences STW for entries that need no work). 0 = no batch + // entry visited yet this FollowReferences call. + jlong _last_visited_batch_tag = 0; + + // Set only by admitStaticFieldRoots(): the seed holder array for that + // sweep holds loaded-class objects (negative-tagged by + // resolveLoadedClasses(), see the *tag_ptr < 0 branch below), and the + // whole point of the sweep is to walk past that holder->class edge into + // each class's own outgoing references - chiefly STATIC_FIELD - which the + // *tag_ptr < 0 check would otherwise stop cold before FollowReferences + // ever gets to report them. Left false everywhere else (expandFrontier()'s + // batching, root enumeration, the whole-heap first pass), where a + // negative-tagged referee must never be descended into. + bool static_field_seed = false; + + // PER-CLASS NON-STATIC QUOTA (admitStaticFieldRoots() only). JVMTI + // reports a class's entire metadata graph through the same + // static_field_seed opening - CONSTANT_POOL (resolved String/Class/ + // MethodHandle/MethodType/CallSite constants), INTERFACE, SUPERCLASS, + // CLASS_LOADER, ... - not just its STATIC_FIELD edges. CONSTANT_POOL + // alone is 5-15x STATIC_FIELD volume per class, systemically. Admitting + // all of them would burn the per-chunk callback/deadline budget on + // non-static-field edges and starve static-field discovery; dropping + // them entirely would exclude a real (if rarer) leak category. Instead, + // STATIC_FIELD edges are always admitted and non-STATIC_FIELD edges from + // a class are admitted up to _class_other_cap per class, then dropped + // for the rest of that class this lap. The cap resets on class + // boundary (detected by referrer tag change), so one fat class cannot + // exhaust the quota for any other. admitStaticFieldRoots() sets + // _class_other_cap from STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS; + // everywhere else these are zero/unused. + jlong _seed_class_tag = 0; // negative tag of the class currently + // being descended (0 before the first + // class edge is seen) + int _class_other_admitted = 0; // non-STATIC_FIELD edges admitted for + // the current class this lap + int _class_other_cap = 0; // per-class cap; 0 disables the quota + // (admit all) when not in seed sweep + // Number of distinct classes entered so far in this chunk's descent + // (incremented on each class-boundary tag change). Diagnostic only - + // the truncation cursor no longer derives from it (a non-HotSpot + // FollowReferences visit order would map the count to the wrong + // original indices; the cursor now redoes the whole chunk). + int _classes_in_chunk_visited = 0; + + // Amortizes tracker->_pass_deadline_ns's OS::nanotime() check (heapReference + // Callback()/heapRootCallback() run once per visited edge/root - checking + // wall-clock on literally every call would add real overhead on a large + // heap) - checked only every 4096th call, local to this ctx so each of + // runPassManualWalk()'s several sub-calls (root enum, static-field sweep, + // expandFrontier(), rotation) starts its own count. + int deadline_check_counter = 0; + + // True while expandFrontier() is walking a batch drawn from + // _priority_expand (a rotation-selected, already-EXPANDED parent) rather + // than the ordinary _pending_expand backlog - see _priority_expand's own + // comment. Newly admitted children inherit the fast lane so the whole + // re-discovered subtree, not just the immediate child, skips the backlog. + bool admit_priority = false; + + // DESCEND-WALK controls (descendFromAnchor()'s calls only; null/0 + // everywhere else, so every gate below is a no-op for the ordinary + // walk phases): + // + // _no_descend_class_tags: exact class tags to neither admit nor descend + // into for the duration of this walk. The fat-metadata classes whose + // graphs would otherwise turn a bounded anchor walk into sweep-scale + // cost (java.lang.ClassLoader from Thread.contextClassLoader reaching + // every loaded class, java.lang.ThreadGroup reaching every thread, + // java.security.ProtectionDomain). Exact tag match only - a SUBCLASS of + // one of these is a distinct class tag and is descended into normally + // (documented limitation: subclass instances are ordinary app objects, + // and an is-a hierarchy walk is not affordable per callback). + // + // _descent_anchor_tag/_anchor_descend_class_tag: when non-zero, the + // ANCHOR object's own outgoing edges are gated to descend only into + // referees of that exact class (walkCandidateThreadLocals() passes + // java.lang.ThreadLocal$ThreadLocalMap - the value type of BOTH of + // Thread's threadLocals and inheritableThreadLocals fields - so the walk + // never enumerates the Thread's other instance fields; the class gate + // is used instead of jvmtiHeapReferenceInfoField.index because the + // index-vs-GetClassFields-order correspondence is a spec subtlety this + // design has no need to depend on). Below the anchor, descent is gated + // by _no_descend_class_tags as above. + static constexpr int NO_DESCEND_CLASS_CAP = 8; + jlong _no_descend_class_tags[NO_DESCEND_CLASS_CAP] = {0}; + int _no_descend_class_tag_count = 0; + jlong _descent_anchor_tag = 0; + jlong _anchor_descend_class_tag = 0; +}; +} // namespace + +jint JNICALL ReferenceChainTracker::heapReferenceCallback( + jvmtiHeapReferenceKind reference_kind, + const jvmtiHeapReferenceInfo *reference_info, jlong class_tag, + jlong referrer_class_tag, jlong size, jlong *tag_ptr, + jlong *referrer_tag_ptr, jint length, void *user_data) { + PassContext *ctx = (PassContext *)user_data; + + if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { + // stopThread() has set this right before pthread_kill()/pthread_join() - + // see that method's own comment. pthread_kill(WAKEUP_SIGNAL) only + // interrupts threadLoop()'s OS::sleep(); it cannot interrupt an + // in-flight JVMTI FollowReferences call, so without this check + // pthread_join() would block until this pass's walk finishes on its own + // - potentially the whole reachable graph, well past any caller's + // shutdown timeout. Treat it exactly like an ordinary budget exhaustion + // (ctx->truncated = true): this pass ends early and the search stays + // non-terminal - fine, since the tracker is shutting down and simply + // never resumes it. + ctx->truncated = true; + return JVMTI_VISIT_ABORT; + } + + if (ctx->tracker->_pass_deadline_ns != 0 && + (++ctx->deadline_check_counter & 0xFFF) == 0 && + OS::nanotime() >= ctx->tracker->_pass_deadline_ns) { + // This pass has run past its wall-clock share (see _pass_deadline_ns's + // own comment) - treat it exactly like ordinary budget exhaustion so it + // ends early without abandoning the search; a later pass re-enumerates + // whatever roots/edges this one didn't get to. + ctx->truncated = true; + return JVMTI_VISIT_ABORT; + } + + // Retention-edge identity for every admission site below: the JVMTI + // heap callback's field ordinal (the JVMTI-SPECIFICATION numbering over + // the referrer's flattened field space - see FrontierEntry:: + // referrer_field_index's own comment) for FIELD/STATIC_FIELD edges, -1 + // otherwise; and the referrer's class tag when the referrer is a + // CLASS OBJECT (root-attached static edges - interior hops get their + // referrer class from the parent entry at chain-reconstruction time, + // so only the parent_tag==0 case needs it recorded here). Captured + // here, before the early-return branches, because a live heap callback + // cannot be replayed after the fact - the same reason FrontierEntry:: + // class_tag is stored at admission. + jint edge_field_index = -1; + if (reference_info != nullptr && + (reference_kind == JVMTI_HEAP_REFERENCE_FIELD || + reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD)) { + edge_field_index = reference_info->field.index; + } + jlong edge_referrer_class_tag = 0; + if (referrer_tag_ptr != nullptr && *referrer_tag_ptr < 0) { + edge_referrer_class_tag = *referrer_tag_ptr; + } + + // NOTE: the old canary/marker-pruning branch (record the chain link for + // an object carrying marker tag MARKER_TAG_BASE - i) is gone: pollWatched + // Targets() stopped assigning marker tags when discovery moved to + // LivenessTracker's leak tags ("_candidate_tags[slot] = 0; // no marker + // tags - using leak tags now"), so no object can ever carry a tag that + // negative (class tags are small negatives from ClassTagAllocator; the + // branch's own frontier->insert() of the negative marker tag could never + // succeed - FrontierTable::insert() rejects tag <= 0 - yet its return + // value was unchecked and the candidate bookkeeping was updated + // regardless). Candidate discovery runs exclusively through the leak-tag + // interception branch below (isLeakTag() -> recordDiscoveredInstance()). + // The _candidate_* arrays and buildCanaryChainEvent() remain only for the + // test seams and the dead-representative reconstruction path in + // pollWatchedTargets(), which this branch used to feed. + // + // Defensive tag_ptr guard: every decode below reads *tag_ptr, and the + // admission path writes back through it. The sibling referrer_tag_ptr is + // null-guarded everywhere (heap-root references have no referrer); + // tag_ptr is treated the same way - a null tag pointer is handled as an + // untagged, non-admittable object rather than dereferenced. + if (tag_ptr == nullptr) { + return 0; + } + + if (*tag_ptr < 0) { + if (ctx->static_field_seed && + reference_kind == JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT && + referrer_tag_ptr != nullptr && *referrer_tag_ptr == 0) { + // admitStaticFieldRoots()'s own holder[i] -> class edge: referrer_tag_ptr + // points at the transient, never-tagged seed array itself (tag 0), not + // at a frontier-admitted parent. Continue the walk into this class's + // own outgoing references - static fields chief among them - instead + // of stopping here; that is the entire purpose of the sweep. The class + // object itself is still never admitted into the frontier (tag_ptr is + // left untouched, so it stays negative). + return JVMTI_VISIT_OBJECTS; + } + // Referee is a class object already tagged negative by + // resolveLoadedClasses() (that pre-pass runs before FollowReferences in + // runPass(), so every loaded class already carries a negative tag by + // this point). Never admit a class object into the frontier as if it + // were an ordinary retained instance, and - outside the + // admitStaticFieldRoots() seed edge handled above - never expand from a + // class's own metadata graph (static fields, superclass, interfaces, + // constant pool, class loader, ...). Out of scope per the design doc's + // non-goals (no field-level/exhaustive paths) and keeps the walk bounded + // to the instance-reachability graph that actually explains "why is + // this object alive". + return 0; + } + if (reference_kind == JVMTI_HEAP_REFERENCE_CLASS || + reference_kind == JVMTI_HEAP_REFERENCE_SYSTEM_CLASS) { + // Definitionally a class by reference_kind (CLASS: "reference from an + // object to its class"; SYSTEM_CLASS: a root reference to a class) even + // if resolveLoadedClasses() failed to resolve/tag this particular one + // (e.g. a transient StringDictionary contention failure) and its tag is + // therefore not yet negative. Same non-goal as above: never expand from + // or admit a class object. + return 0; + } + + if (ctx->truncated) { + // Defensive: FollowReferences should already have stopped delivering + // callbacks after a JVMTI_VISIT_ABORT return below; this just avoids + // doing further work if one more callback arrives anyway. + return JVMTI_VISIT_ABORT; + } + + jlong parent_tag = 0; + u32 depth = 0; + if (referrer_tag_ptr != nullptr) { + jlong rtag = *referrer_tag_ptr; + if (rtag > 0) { + FrontierEntry parent{}; + if (ctx->frontier->lookup(rtag, &parent)) { + parent_tag = rtag; + depth = parent.depth + 1; + } + // lookup() failing for a positive rtag should not happen - a referrer + // must already be one of our tagged frontier objects for its own + // outgoing edges to be traversed at all (FollowReferences only + // explores past an object this callback returned JVMTI_VISIT_OBJECTS + // for) - but fall back to root-like (parent_tag=0/depth=0) rather + // than corrupt the chain if it ever does. + } + // rtag < 0: referrer is a pre-tagged class object (e.g. a static field + // holding this reference) - treated as root-like rather than attributed + // to a parent hop, since class objects are never admitted as frontier + // entries and so have no depth/parent_tag of their own (see the + // *tag_ptr < 0 check above). rtag == 0: referrer not yet tagged, should + // not happen for the same reason noted above. + } + // referrer_tag_ptr == nullptr: a heap-root reference (JNI global, thread + // stack local/JNI local, monitor, thread, system class, ...) - parent_tag + // and depth stay 0. + + if (depth >= (u32)ctx->hop_cap) { + // Hop cap: do not admit this object into the frontier, and do not + // expand further from it - enforced here rather than + // discovering-then-discarding. + return 0; + } + + if (ctx->static_field_seed && referrer_tag_ptr != nullptr && + *referrer_tag_ptr < 0) { + // Referrer is the class object opened by the static_field_seed branch + // above. JVMTI reports that class's entire metadata reference graph + // through this same opening, not just its static fields - CONSTANT_POOL + // (resolved String/Class/MethodHandle/MethodType/CallSite constants), + // INTERFACE, SUPERCLASS, CLASS_LOADER, ... Those are real reachability + // edges, just lower-priority for this static-field-root sweep than + // STATIC_FIELD. Admit STATIC_FIELD unconditionally; admit non-STATIC_FIELD + // up to the per-class cap (PassContext::_class_other_cap) so one fat + // class cannot starve the rest, then drop further non-static edges for + // this class this lap. Track the current class tag for + // admitStaticFieldRoots()'s resumable cursor (see that method's own + // comment) and reset the cap counter on class boundary. + if (*referrer_tag_ptr != ctx->_seed_class_tag) { + ctx->_seed_class_tag = *referrer_tag_ptr; + ctx->_class_other_admitted = 0; + ctx->_classes_in_chunk_visited++; + } + if (reference_kind != JVMTI_HEAP_REFERENCE_STATIC_FIELD) { + if (ctx->_class_other_cap > 0 && + ctx->_class_other_admitted >= ctx->_class_other_cap) { + // Quota exhausted for this class - drop the edge. Count every + // drop, and count the first drop for this class separately so + // the two counters together distinguish "a few fat outlier + // classes dropping many edges" from "systematic drops across + // almost all classes" (cap too low). + Counters::increment(REFERENCE_CHAIN_STATIC_SWEEP_NON_STATIC_DROPPED); + if (ctx->_class_other_admitted == ctx->_class_other_cap) { + Counters::increment(REFERENCE_CHAIN_STATIC_SWEEP_CLASSES_CAPPED); + } + return 0; + } + ctx->_class_other_admitted++; + } + } + + // Leak tag: this object was directly tagged by LivenessTracker's + // tagLeakInstances() because it's a tracked leaking object. Convert + // the leak tag to a frontier tag so the BFS can build its chain, and + // store the leak tag in the frontier entry for correlation with + // HeapLiveObject events. + if (isLeakTag(*tag_ptr)) { + jlong leak_tag = *tag_ptr; + // Allocate a frontier tag for this object + jlong frontier_tag = ctx->tracker->nextTag(); + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + u8 root_kind = parent_tag == 0 ? (u8)reference_kind : 0; + if (ctx->frontier->insert(frontier_tag, parent_tag, referrer_klass, + depth, FrontierEntryState::FRONTIER, + root_kind, class_tag, edge_field_index, + (u8)reference_kind, + parent_tag == 0 ? edge_referrer_class_tag : 0)) { + // Store the leak tag in the frontier entry + ctx->frontier->setLeakTag(frontier_tag, leak_tag); + *tag_ptr = frontier_tag; + ctx->edges_admitted++; + TEST_LOG("ReferenceChainTracker::heapReferenceCallback leak-tag " + "intercepted: leak_tag=%lld -> frontier_tag=%lld depth=%u " + "parent_tag=%lld", + (long long)leak_tag, (long long)frontier_tag, depth, + (long long)parent_tag); + ctx->tracker->trackLeakAccumulation(ctx->frontier, class_tag, + parent_tag, frontier_tag); + // Index maintenance: a leak-tagged object admitted root-attached by + // a durable root edge (e.g. a static field directly holding a tagged + // chunk) is the highest-priority anchor tier (leak_tag != 0). + if (parent_tag == 0 && + (root_kind == (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || + root_kind == (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL)) { + ctx->tracker->addToStaticAnchorIndex(frontier_tag, class_tag, + root_kind); + } + // Auto-mark: record this as a discovered instance, with eviction + // rights over uncorrelated noise slots (see recordDiscoveredInstance). + if (ctx->tracker->_candidate_count > 0) { + u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); + ctx->tracker->recordDiscoveredInstance(klass_id, frontier_tag, true); + } + } else { + // Frontier cap hit + ctx->truncated = true; + ctx->frontier_cap_hit = true; + return JVMTI_VISIT_ABORT; + } + return JVMTI_VISIT_OBJECTS; + } + + // DESCEND-WALK GATES (no-ops on every ordinary walk - the PassContext + // fields below are zero-initialized and only descendFromAnchor() sets + // them). Deliberately placed AFTER the leak-tag interception branch + // above: a leak-tagged instance of a no-descend class (e.g. a leaking + // ClassLoader - a classic leak category) must still be intercepted and + // correlated; only its own subtree is not descended into. Returning 0 + // skips admission AND descent for this edge, which also skips the + // improveChain/root-upgrade branches below - correct for this walk: a + // descend walk's purpose is reaching tagged instances below the anchor, + // not re-attributing metadata objects the ordinary BFS already owns. + if (ctx->_no_descend_class_tag_count > 0) { + for (int i = 0; i < ctx->_no_descend_class_tag_count; i++) { + if (ctx->_no_descend_class_tags[i] == class_tag) { + return 0; + } + } + } + if (ctx->_descent_anchor_tag != 0 && ctx->_anchor_descend_class_tag != 0 && + referrer_tag_ptr != nullptr && + *referrer_tag_ptr == ctx->_descent_anchor_tag && + class_tag != ctx->_anchor_descend_class_tag) { + // This descend walk's ANCHOR object's own edge, and the referee is not + // the gate class (see PassContext::_anchor_descend_class_tag's own + // comment - e.g. walkCandidateThreadLocals() walks ONLY the Thread's + // ThreadLocalMap edges, never enumerating the Thread's other fields). + return 0; + } + + if (*tag_ptr == 0) { + // First time this object is visited in this pass. + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + // reference_kind describes this admitting edge; only meaningful for a + // root-attached entry (parent_tag == 0) - see FrontierEntry::root_kind's + // own comment for why a non-root entry's edge kind is not recorded. + u8 root_kind = parent_tag == 0 ? (u8)reference_kind : 0; + ReferenceChainTracker::AdmitResult result = ctx->tracker->admitObject( + ctx->frontier, ctx->hop_cap, ctx->budget, &ctx->edges_admitted, + tag_ptr, parent_tag, referrer_klass, depth, root_kind, class_tag, + ctx->admit_priority, edge_field_index, (u8)reference_kind, + parent_tag == 0 ? edge_referrer_class_tag : 0); + switch (result) { + case ReferenceChainTracker::AdmitResult::BUDGET_EXHAUSTED: + ctx->truncated = true; + return JVMTI_VISIT_ABORT; + case ReferenceChainTracker::AdmitResult::FRONTIER_CAP_HIT: + // Frontier-size cap hit (FrontierTable::insert() returned false + // without partially writing) - stop admitting new entries and report + // the truncation (design doc: "stop admitting new entries ... report + // it"), rather than silently dropping this object and continuing. + // Distinct from ordinary budget exhaustion above - runPass() keeps + // the search RUNNING on this, deferring to the no-progress detector + // to abandon only if the frontier then stops growing (see runPass()'s + // frontier_cap_hit handling). + ctx->truncated = true; + ctx->frontier_cap_hit = true; + return JVMTI_VISIT_ABORT; + default: + // ADMITTED, or HOP_CAP/ALREADY_ADMITTED (neither reachable here: the + // hop-cap check above already returned before this branch, and + // *tag_ptr == 0 rules out ALREADY_ADMITTED) - nothing further to do. + break; + } + // Index maintenance: track root-attached durable anchors for O(anchors) + // collector iteration instead of O(frontier_size) table scan. class_tag + // is the anchor object's OWN class tag (the callback's class_tag param + // describes the referee, i.e. the object being admitted here) - kept so + // the collector can tier by class shape without JNI. + if (result == ReferenceChainTracker::AdmitResult::ADMITTED && + parent_tag == 0) { + ctx->tracker->addToStaticAnchorIndex(*tag_ptr, class_tag, root_kind); + } + // Auto-mark: if this object's class matches a watched leak class, + // record its frontier tag so pollWatchedTargets() can build a chain + // event for it. A leaking class typically has many live instances, + // and each one's reference chain is independently useful — the + // pre-tagged representative is just one sample, and its representative + // may change (LRU-evicted) between polls. Recording all discovered + // instances ensures we emit chain events for all of them, not just + // whichever single object happened to be the representative when the + // canary slot was first filled. See _candidate_discovered_tags's own + // comment. + if (result == ReferenceChainTracker::AdmitResult::ADMITTED && + ctx->tracker->_candidate_count > 0) { + u32 klass_id = ctx->tracker->classTags()->resolve(class_tag); + if (klass_id == 0) { + // class_tag not in _class_tags - either class map rotated + // (resolveLoadedClasses hasn't re-resolved yet) or this class + // was never tagged. Log once per pass to diagnose class-map + // rotation issues. + TEST_LOG("ReferenceChainTracker::auto-mark class_tag=%lld " + "unresolved (not in _class_tags)", + (long long)class_tag); + } else { + bool matched = false; + for (int s = 0; s < ctx->tracker->_candidate_count; s++) { + if (ctx->tracker->_candidate_klass_ids[s] == klass_id) { + matched = true; + ctx->tracker->recordDiscoveredInstance(klass_id, *tag_ptr, + false); + break; + } + } + if (!matched && klass_id != 0) { + // klass_id resolved but doesn't match any candidate - likely + // class map rotation made candidate klass_ids stale + TEST_LOG("ReferenceChainTracker::auto-mark klass_id=%u " + "resolved but no candidate match (candidates=[%u,%u,%u,%u,%u])", + klass_id, + ctx->tracker->_candidate_count > 0 ? ctx->tracker->_candidate_klass_ids[0] : 0, + ctx->tracker->_candidate_count > 1 ? ctx->tracker->_candidate_klass_ids[1] : 0, + ctx->tracker->_candidate_count > 2 ? ctx->tracker->_candidate_klass_ids[2] : 0, + ctx->tracker->_candidate_count > 3 ? ctx->tracker->_candidate_klass_ids[3] : 0, + ctx->tracker->_candidate_count > 4 ? ctx->tracker->_candidate_klass_ids[4] : 0); + } + } + } + } else if (*tag_ptr > 0) { + // Already-tagged object reached via a new edge. This arm - NOT the + // first-admission block above - is where an already-admitted entry's + // shape can be corrected: improveChain/reparentToDurableRoot for a + // deeper/equal-durable path, maybeUpgradeRootAttachedRootKind for a + // new root-like edge. These branches were originally nested INSIDE + // the *tag_ptr == 0 block (misplaced by 57aec4895, whose own message + // says "improveChain needs to run when *tag_ptr != 0"), where they + // were dead code for their stated purpose: a freshly-admitted entry + // carries exactly this edge's (parent_tag, depth), so both improve- + // Chain's depth> check and the durability upgrade's strict-> check + // are guaranteed no-ops there. On the pod this silently disabled + // every already-admitted re-attribution: the static sweep's edge onto + // a holder born chain-attached could never re-root it + // (find-anchor-holder-eviction). + if (parent_tag != 0) { + // This new path is deeper - replace the shallow root-attached entry + // with the deeper chain-attached entry. This fixes the "depth=1 chain + // with no holder" problem: an object first admitted as a JNI-local + // root (parent_tag == 0) gets its frontier entry improved when the + // static-field → ... → object path reaches it. + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + // Pre-read the CURRENT shape: if improveChain() below succeeds, this + // root-attached durable entry is about to be replaced with a deeper + // chain-attached one - i.e. it is leaving the population + // collectStaticFieldAnchorsForRotation() can select, at exactly this + // moment. That eviction (find-anchor-holder-eviction) needs a B' + // at-risk push fired HERE, not only from the sweep's static-edge + // site: the sweep gate re-laps only while the class count is in flux + // (see admitStaticFieldRoots' gate in runPassManualWalk), so a + // post-lap demotion in a stable-class JVM would otherwise never see + // another static edge onto this entry. + FrontierEntry pre_improve_entry{}; + bool was_root_attached_durable = + ctx->frontier->lookup(*tag_ptr, &pre_improve_entry) && + pre_improve_entry.parent_tag == 0 && + rootKindDurability(pre_improve_entry.root_kind) >= 2; + if (ctx->frontier->improveChain(*tag_ptr, parent_tag, referrer_klass, + depth, 0, edge_field_index, + (u8)reference_kind)) { + // Chain was improved — invalidate any cached chain for this tag + // so pollWatchedTargets rebuilds it with the deeper path. + // Deferred (see deferResolvedChainInvalidation): this runs inside + // the FollowReferences STW pause; taking _resolved_chains_lock + // here can block the walk behind drainPendingChainEvents()'s + // full-cache copy on the JFR dump thread. + ctx->tracker->deferResolvedChainInvalidation(*tag_ptr); + if (was_root_attached_durable) { + // Demotion push (B'): the replaced entry's static/JNI-global + // attribution was its only anchor-tier eligibility, and it is + // gone now. rootKindDurability() >= 2 is exactly the durable set + // the collector selects (STATIC_FIELD, JNI_GLOBAL); SYSTEM_CLASS + // scores 3 too but a class object is never admitted as a frontier + // entry, so it cannot appear here. + ctx->tracker->pushAtRiskStaticAnchor( + *tag_ptr, ctx->tracker->classTags()->resolve(class_tag)); + } + } else if (ctx->frontier->reparentToDurableRoot( + *tag_ptr, parent_tag, referrer_klass, edge_field_index, + (u8)reference_kind)) { + // Equal-depth re-parent from a transient root to a durable one + // (improveChain() cannot express it - see its declaration) - same + // cache invalidation so the rebuilt chain uses the durable root. + // Deferred: same STW-lock rationale as the improveChain branch. + ctx->tracker->deferResolvedChainInvalidation(*tag_ptr); + } + } else { + // Already-admitted entry reached via a NEW root-like edge + // (parent_tag == 0): the static-field sweep's class -> field edge + // reports the class as the referrer with a negative tag, which the + // rtag < 0 branch above treats as root-like (class objects are never + // frontier entries), and heap-root references arrive here with + // referrer_tag_ptr == nullptr. Without this, an entry first admitted + // through a stack local keeps its transient classification forever + // even after a later static-field sweep proves the same object is + // the direct value of a static field - exactly the durable-root + // discovery maybeUpgradeRootAttachedRootKind() exists for (same + // tie-break heapRootCallback() applies on its own ALREADY_ADMITTED + // case), so reuse it: upgrade only when this edge's kind is strictly + // more durable, and drop any cached chain so it is rebuilt with the + // upgraded root kind. + if (ctx->tracker->maybeUpgradeRootAttachedRootKind(ctx->frontier, + *tag_ptr, + (u8)reference_kind)) { + // Deferred: same STW-lock rationale as the improveChain branch. + ctx->tracker->deferResolvedChainInvalidation(*tag_ptr); + } else if (reference_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) { + // The upgrade refused (maybeUpgradeRootAttachedRootKind returns + // false for parent_tag != 0 by design), so this STATIC_FIELD edge + // just proved an at-risk static attachment the anchor tier's + // parent_tag == 0 filter can never see: a holder already admitted + // as a non-root child (find-anchor-holder-eviction). Feed it to + // _static_anchor_fifo (B', see its declaration comment) so the + // static-anchor walk drains it ahead of the root-attached cohort. + // Only the sweep emits STATIC_FIELD edges onto already-tagged + // entries: the edge's referrer is the class object, and the BFS/ + // descend walks never expand classes (the tag < 0 and CLASS-kind + // early returns above), so no other walk can flood the FIFO. + FrontierEntry entry{}; + if (ctx->frontier->lookup(*tag_ptr, &entry) && + entry.parent_tag != 0) { + ctx->tracker->pushAtRiskStaticAnchor( + *tag_ptr, ctx->tracker->classTags()->resolve(class_tag)); + } + } + } + } + + if (ctx->batch_tags != nullptr) { + // ARRAY-HOLDER BATCHING one-hop descent control (see PassContext:: + // batch_tags). Descend only into this pass's boundary objects so the + // single FollowReferences(holder_array) call expands exactly one hop: + // a boundary object yields its direct children (which get tagged above), + // but those children are not themselves descended into, and any + // already-expanded object from a prior pass is skipped rather than + // re-traversed. + jlong my_tag = *tag_ptr; + if (my_tag > 0 && ctx->batch_tags->count(my_tag) != 0) { + // Track this batch entry as visited for the rolling resume cursor + // (see _last_visited_batch_tag's own comment). + ctx->_last_visited_batch_tag = my_tag; + return JVMTI_VISIT_OBJECTS; + } + return 0; + } + + return JVMTI_VISIT_OBJECTS; +} + +ReferenceChainTracker::AdmitResult ReferenceChainTracker::admitObject( + FrontierTable *frontier, int hop_cap, int budget, int *edges_admitted, + jlong *tag_ptr, jlong parent_tag, u32 referrer_klass, u32 depth, + u8 root_kind, jlong class_tag, bool priority, + jint edge_field_index, u8 edge_kind, jlong edge_referrer_class_tag) { + // edge_* default-declared in the header; heapRootCallback() passes the + // defaults (a root reference is not a field edge) unchanged. + // Null guard, same rationale as heapReferenceCallback()'s own: the + // admit path WRITES the new tag back through tag_ptr below. + if (tag_ptr == nullptr) { + return AdmitResult::ALREADY_ADMITTED; + } + if (*tag_ptr != 0) { + return AdmitResult::ALREADY_ADMITTED; + } + if (depth >= (u32)hop_cap) { + return AdmitResult::HOP_CAP; + } + if (*edges_admitted >= budget) { + return AdmitResult::BUDGET_EXHAUSTED; + } + jlong tag = nextTag(); + if (!frontier->insert(tag, parent_tag, referrer_klass, depth, + FrontierEntryState::FRONTIER, root_kind, class_tag, + edge_field_index, edge_kind, edge_referrer_class_tag)) { + return AdmitResult::FRONTIER_CAP_HIT; + } + *tag_ptr = tag; + (*edges_admitted)++; + // Queue for expandFrontier()/markAllFrontierExpanded() - see + // _pending_expand's/_priority_expand's own declaration comments for why + // this replaces a scan over the admitted range, and for why a + // rotation-discovered child (priority=true) skips the ordinary backlog. + if (priority && _priority_expand.size() < PRIORITY_EXPAND_CAP) { + _priority_expand.push_back(tag); + _priority_expand_set.insert(tag); + } else { + // Priority lane full: the rotation backpressure falls back to the + // ordinary backlog rather than silently dropping the re-discovered + // subtree (see PRIORITY_EXPAND_CAP's own comment). + _pending_expand.push_back(tag); + } + trackLeakAccumulation(frontier, class_tag, parent_tag, tag); + return AdmitResult::ADMITTED; +} + +void ReferenceChainTracker::trackLeakAccumulation(FrontierTable *frontier, + jlong class_tag, + jlong parent_tag, + jlong tag) { + // Cheapest checks first: no klass_id is currently watched (the common + // case before hasLeakSignal() has ever fired - see + // _watched_leak_klass_ids' own comment), or this admission has no real + // parent to attribute to (a root-attached entry - nothing to aggregate + // by, since the "container" concept this tracks is specifically about a + // PARENT object's field holding the leaf, not the leaf itself being + // root-attached). + if (_watched_leak_klass_count <= 0 || parent_tag == 0 || class_tag == 0) { + return; + } + // (u32) truncation matches _watched_leak_klass_ids' own storage (see that + // field's comment) - class tags are small, negative, sequentially-minted + // values in practice (ClassTagAllocator::next()), so this never actually + // loses distinguishing information; it just keeps the comparison and the + // signature-key packing below in the same 32-bit space both already used + // for the (superseded) classMap-id scheme. + u32 truncated_class_tag = (u32)class_tag; + bool watched = false; + for (int i = 0; i < _watched_leak_klass_count; i++) { + if (_watched_leak_klass_ids[i] == truncated_class_tag) { + watched = true; + break; + } + } + if (!watched) { + return; + } + FrontierEntry parent_entry{}; + if (!frontier->lookup(parent_tag, &parent_entry) || + parent_entry.class_tag == 0) { + // Parent since pruned/dead between its own admission and this child's, + // or admitted before this field existed on it (should not happen in + // practice - class_tag is set at every admission - but a stale/unknown + // parent identity is not something to attribute this observation to + // either way. + return; + } + u64 key = leakSignatureKey(truncated_class_tag, (u32)parent_entry.class_tag); + _leak_signature_totals[key]++; + auto it = _leak_parent_fanout.find(parent_tag); + if (it == _leak_parent_fanout.end()) { + TEST_LOG("ReferenceChainTracker::trackLeakAccumulation fanout-insert " + "parent_tag=%lld parent_class_tag=%lld child_class_tag=%lld", + (long long)parent_tag, (long long)parent_entry.class_tag, + (long long)class_tag); + _leak_parent_fanout.emplace(parent_tag, LeakParentFanoutEntry{key, 1}); + } else { + // The signature key for a given parent_tag is fixed once recorded + // (parent_entry.class_tag never changes once admitted; the LEAF side of + // the key is fixed by which klass_id is currently watched at the time + // of THIS call, which could in principle differ between two children of + // the same parent if _watched_leak_klass_ids itself changed between + // them - overwrite rather than accumulate under a stale key in that + // case, since the stored signature_key should always reflect the most + // recently observed watched klass_id for this parent). + it->second.signature_key = key; + it->second.fanout++; + } + // ANCESTOR FANOUT: the direct parent is not necessarily the part of the + // holder chain that STAYS LIVE. A container that replaces its internals + // (the canonical unmaintained-singleton leak: a growing ArrayList swaps + // elementData on growth, a HashMap resizes its table) leaves the watched + // instances' direct parents dead - observed live: the fanout filled with + // old backing arrays while the live holder was never re-walked, its new + // internals never admitted, and zero tagged chunks ever intercepted. + // The ancestors up to the root ARE the durable holders, so record every + // hop of the holder chain, not just the last one. Bounded: this only runs + // for watched-klass admissions (rare - leak-candidate classes only), and + // the walk stops at the root-attached entry (parent_tag == 0), typically + // a handful of lookups. + jlong ancestor = parent_entry.parent_tag; + int hops = 0; + while (ancestor != 0 && hops++ < _hop_cap) { + FrontierEntry ancestor_entry{}; + if (!_frontier->lookup(ancestor, &ancestor_entry)) { + break; + } + if (_leak_parent_fanout.find(ancestor) == _leak_parent_fanout.end()) { + _leak_parent_fanout.emplace(ancestor, LeakParentFanoutEntry{key, 1}); + } + if (ancestor_entry.parent_tag == 0) { + break; // root-attached: the holder chain ends here + } + ancestor = ancestor_entry.parent_tag; + } +} + +void ReferenceChainTracker::seedLeakAccumulationForNewlyWatchedKlass( + u32 klass_id) { + if (_frontier == nullptr) { + // pollWatchedTargets() can run before the first pass has ever created + // the frontier table - nothing to seed from yet. + return; + } + int table_size = _frontier->size(); + if (table_size <= 0) { + return; + } + // Inlines trackLeakAccumulation()'s own signature/fanout update logic + // (rather than calling it per matching entry) deliberately: this whole + // scan already holds _frontier's shared lock for its duration (matching + // collectStaleExpandedEntriesForRotation()'s own lockShared() rationale - + // a per-tag SpinLock acquisition would double the cost of this O(table_size) + // sweep), and trackLeakAccumulation() takes that same lock itself via + // frontier->lookup() - calling it from inside an already-held shared + // section would risk a reentrant-lock deadlock if a writer is ever + // concurrently pending, so this uses lookupLocked() throughout instead. + // Compares against (u32) FrontierEntry::class_tag - see that field's own + // comment for why this, and not referrer_klass, is the stable identifier + // klass_id (itself a truncated class_tag - _watched_leak_klass_ids' own + // comment) can actually be matched against. + _frontier->withSharedLock([&](const FrontierTable *frontier) { + for (jlong tag = 1; tag <= table_size; tag++) { + FrontierEntry entry{}; + if (!frontier->lookupLocked(tag, &entry) || + entry.state != FrontierEntryState::EXPANDED || + entry.parent_tag == 0 || (u32)entry.class_tag != klass_id) { + continue; + } + FrontierEntry parent_entry{}; + if (!frontier->lookupLocked(entry.parent_tag, &parent_entry) || + parent_entry.class_tag == 0) { + continue; + } + u64 key = leakSignatureKey(klass_id, (u32)parent_entry.class_tag); + _leak_signature_totals[key]++; + auto it = _leak_parent_fanout.find(entry.parent_tag); + if (it == _leak_parent_fanout.end()) { + _leak_parent_fanout.emplace(entry.parent_tag, + LeakParentFanoutEntry{key, 1}); + } else { + it->second.signature_key = key; + it->second.fanout++; + } + } + }); +} + +bool ReferenceChainTracker::maybeUpgradeRootAttachedRootKind( + FrontierTable *frontier, jlong tag, u8 new_root_kind) { + FrontierEntry entry{}; + if (!frontier->lookup(tag, &entry)) { + return false; + } + if (entry.parent_tag != 0) { + // Not root-attached - per this phase's option (a) resolution of the + // parent_tag==0/root_kind invariant conflict (referenceChains.h's + // FrontierEntry::root_kind comment), only a root-context update may ever + // write a non-zero root_kind, and only onto an entry that is already + // root-attached. An object that happens to also be a genuine GC root but + // was first discovered as a non-root child (e.g. via frontier + // expansion) keeps its original, non-root attribution - a known, + // documented limitation rather than an attempt to retroactively flip + // parent_tag to 0, which reconstructChain()'s parent-link walk does not + // support. + return false; + } + if (rootKindDurability(new_root_kind) <= rootKindDurability(entry.root_kind)) { + return false; + } + frontier->updateRootKind(tag, new_root_kind); + addToStaticAnchorIndex(tag, entry.class_tag, new_root_kind); + return true; +} + +std::vector +ReferenceChainTracker::collectStaleRootKindEntriesForRotation( + int max_count) { + std::vector selected; + int table_size = _frontier->size(); + if (max_count <= 0 || table_size <= 0) { + return selected; + } + if (_root_kind_rotation_cursor <= 0 || + _root_kind_rotation_cursor > table_size) { + _root_kind_rotation_cursor = 1; + } + + // Held for the whole sweep below (potentially wrapping all the way around + // table_size) rather than once per tag via lookup() - the same rationale + // as collectStaleExpandedEntriesForRotation()'s own lockShared() use: a + // per-tag SpinLock acquisition would double this scan's cost under a large + // frontier table. + jlong start_tag = _root_kind_rotation_cursor; + jlong tag = start_tag; + _frontier->withSharedLock([&](const FrontierTable *frontier) { + do { + FrontierEntry entry{}; + if (frontier->lookupLocked(tag, &entry) && + entry.state == FrontierEntryState::EXPANDED && + entry.parent_tag == 0 && isTransientRootKind(entry.root_kind) && + !isQueuedForRotation(tag) && + _priority_expand.size() < PRIORITY_EXPAND_CAP) { + selected.push_back(tag); + _priority_expand.push_back(tag); + _priority_expand_set.insert(tag); + if ((int)selected.size() >= max_count) { + tag = tag % table_size + 1; + break; + } + } + tag = tag % table_size + 1; + } while (tag != start_tag); + }); + + _root_kind_rotation_cursor = tag; + return selected; +} + +std::vector +ReferenceChainTracker::collectStaleExpandedEntriesForRotation( + int max_count) { + std::vector selected; + int table_size = _frontier->size(); + if (max_count <= 0 || table_size <= 0) { + return selected; + } + // LEAK-PARENT PRIORITY, FAIR-SHARED WITH THE BLIND LAP: _leak_parent_fanout + // knows the EXPANDED parents that actually lead to watched leak-klass + // children - re-walking one of those re-sees its current children + // (improveChain() upgrades children first admitted via a shallower path, + // leak-tag interception for the tagged ones) and catches elements added + // since its expansion, which is exactly the mutation this rotation exists + // to observe. The fanout is orders of magnitude smaller than the table; + // select from it first (rotating via _leak_parent_rotation_cursor for + // coverage), up to HALF the budget (ceil) - then the blind table lap below + // fills the remainder. + // + // Why capped at half rather than fanout-first-until-exhausted (the + // original design, observed broken live): the fanout only ever contains + // parents of watched instances ALREADY ADMITTED as their direct children - + // and for a container that REPLACES its internals (the canonical + // unmaintained-singleton case: a growing ArrayList swaps elementData on + // growth), the watched instances' direct parents are the OLD, now-dead + // backing arrays, while the live holder's new internals are never in the + // fanout at all (the holder's own direct children are non-watched + // container internals). Re-walking the LIVE holder is what admits each + // new backing array; only the blind lap selects an arbitrary EXPANDED + // holder. With an unbounded fanout-first policy and a fanout grown to + // ~11k entries, the fanout filled ALL 256 selections every pass + // (observed live: rotation edges admitted in only 4 of 206 passes, the + // sink's resized backing arrays never admitted, zero interceptions) and + // the lap never ran - the exact starvation this rotation was built to + // prevent, reproduced one level down. A half/half split guarantees both + // tiers make progress every pass. + // + // FANOUT HYGIENE: entries whose parent no longer resolves in the + // frontier (pruned: dead object, search-restart wipe) can never be + // re-walked again, yet accumulate forever without this erase - observed + // live as an 11k-entry fanout of overwhelmingly-dead old backing arrays, + // which both bloats this scan and makes _leak_parent_rotation_cursor's + // lap arithmetic cover mostly corpses. Entries that exist but are not + // EXPANDED yet (still pending expansion) are kept - their children have + // not even been seen once. + if (!_leak_parent_fanout.empty() && + _priority_expand.size() < PRIORITY_EXPAND_CAP) { + int fanout_budget = (max_count + 1) / 2; + size_t fanout_size = _leak_parent_fanout.size(); + u64 skip = _leak_parent_rotation_cursor % fanout_size; + auto it = _leak_parent_fanout.begin(); + while (it != _leak_parent_fanout.end()) { + if ((int)selected.size() >= fanout_budget || + _priority_expand.size() >= PRIORITY_EXPAND_CAP) { + break; + } + if (skip > 0) { + skip--; + ++it; + continue; + } + jlong parent_tag = it->first; + if (isQueuedForRotation(parent_tag)) { + ++it; + continue; + } + FrontierEntry entry{}; + // Dead parent: either the frontier slot is gone entirely, or it was + // clear()'d (dead object / restart wipe) - clear() marks the slot + // ABANDONED rather than removing it, so both conditions must erase + // (tags are never reused within a search and the fanout is wiped on + // restart, so an ABANDONED parent can never come back to life). + if (!_frontier->lookup(parent_tag, &entry) || + entry.state == FrontierEntryState::ABANDONED) { + it = _leak_parent_fanout.erase(it); + continue; + } + if (entry.state != FrontierEntryState::EXPANDED) { + ++it; + continue; + } + selected.push_back(parent_tag); + _priority_expand.push_back(parent_tag); + _priority_expand_set.insert(parent_tag); + ++it; + } + _leak_parent_rotation_cursor += selected.size() + 1; + if ((int)selected.size() >= max_count) { + // Budget exhausted by the fanout alone (only possible for + // max_count == 1, where the fanout's ceil-half share is the whole + // budget) - fanout-priority preserved, and the lap below has nothing + // left to do this pass. + return selected; + } + } + if (_stale_expanded_rotation_cursor <= 0 || + _stale_expanded_rotation_cursor > table_size) { + _stale_expanded_rotation_cursor = 1; + } + // Resume scanning from _stale_expanded_rotation_cursor rather than always + // restarting at tag 1: a frontier table can accumulate far more than + // max_count entries that are EXPANDED and stay that way forever + // (long-lived infrastructure objects - caches, maps, bootstrap classes). + // An always-from-1 scan would let that low-tag population fill this + // sweep's entire cap on every single call, permanently starving any + // EXPANDED entry with a higher tag (e.g. a static field's collection, + // admitted only once its class loads well after startup) of ever being + // re-queued. A wrapping cursor, like collectStaleRootKindEntriesForRotation() + // above already uses, guarantees every entry gets a turn within + // ceil(table_size / max_count) calls instead of never. + // + // This scan's own EXPANDED criterion is a strict superset of + // collectStaleRootKindEntriesForRotation()'s (which additionally requires + // parent_tag == 0 and a transient root_kind), and that function always + // runs first within the same pass and pushes its picks onto + // _priority_expand before this one runs - so without a check here, a tag + // it already selected would be pushed a second time, and + // expandFrontier() re-expands each deque entry as its own independent + // unit of work. isQueuedForRotation() also covers any entries still + // sitting there from a prior pass's truncated batch (expandFrontier() + // leaves those at the front of the queue for a later retry rather than + // popping them). + // Held for the whole scan below instead of once per tag via lookup() - a + // per-tag SpinLock acquisition/release would double the cost of this + // O(table_size) sweep under a large frontier table (the exact scenario - + // tens of thousands of entries - this rotation mechanism targets). + // + // A sparse stretch of non-EXPANDED/already-queued tags could make this + // scan run long chasing max_count with no wall-clock bound of its own - + // unlike heapReferenceCallback()'s per-edge check, this scan isn't itself + // a JVMTI/STW call, but it still steals from the same _pass_deadline_ns + // window the actual walk needs (see that field's own comment). Amortized + // the same way heapReferenceCallback() amortizes its own check: an + // OS::nanotime() call every iteration would + // itself be a meaningful fraction of this loop's per-tag cost. + int deadline_check_counter = 0; + jlong start_tag = _stale_expanded_rotation_cursor; + jlong tag = start_tag; + _frontier->withSharedLock([&](const FrontierTable *frontier) { + do { + if (_pass_deadline_ns != 0 && + (++deadline_check_counter & 0xFFF) == 0 && + OS::nanotime() >= _pass_deadline_ns) { + // Ran past this pass's wall-clock share - stop scanning with + // whatever was already selected (possibly none) and resume from + // here next call. The wrapping cursor already tolerates a call that + // selects fewer than max_count, so this composes without any + // special-casing. + break; + } + FrontierEntry entry{}; + if (frontier->lookupLocked(tag, &entry) && + entry.state == FrontierEntryState::EXPANDED && + !isQueuedForRotation(tag) && + _priority_expand.size() < PRIORITY_EXPAND_CAP) { + selected.push_back(tag); + _priority_expand.push_back(tag); + _priority_expand_set.insert(tag); + if ((int)selected.size() >= max_count) { + tag = tag % table_size + 1; + break; + } + } + tag = tag % table_size + 1; + } while (tag != start_tag); + }); + _stale_expanded_rotation_cursor = tag; + return selected; +} + +// Bounded rotating re-expansion targeting the accumulation point of a +// klass LivenessTracker has flagged as growing - the design's actual +// targeted tier, replacing an earlier structural (depth + root-durability + +// class-shape) heuristic that measurement against a real classpath showed +// selects far too much of the reachable graph to fit in a small budget (see +// git history and doc/temp/ investigation notes). This design instead uses +// the one signal that CAN distinguish "the specific container that is +// leaking" from "the many unrelated objects that happen to hold instances +// of a common leaf class" without needing a full dominator-tree/retained- +// size computation (a wider web search into how heap analysis tools solve +// this - Eclipse MAT's accumulation-point/big-drop-in-dominator-tree +// heuristic, Cork's class-level points-from summary diffed across GCs, +// LeakBot's rank-cheaply-then-track-only-the-winners discipline - converged +// on the same two-tier shape implemented here): +// +// Tier 1 (class-level, cheap, aggregated at admission time by +// trackLeakAccumulation() into _leak_signature_totals/_leak_parent_fanout - +// no full-table scan): rank (leaf_klass_id, parent_class_id) signatures by +// growth since the previous pass (current total minus the snapshot rolled +// forward at the end of that pass), Cork-style. This collapses "thousands +// of objects holding a common leaf class" into a handful of signatures - a +// legitimate cache class that merely holds MANY instances, but not a +// GROWING number of them pass over pass, never wins here, regardless of its +// absolute size. +// +// Tier 2 (spent only within the winning signature): rank the concrete +// parent objects contributing to it by their own fanout of the flagged +// leaf klass_id - the highest-fanout parent is the one whose already- +// EXPANDED state is most likely stale (i.e. its own children were admitted +// once and it has since accumulated more that were never observed), so it +// is the one worth spending this pass's rotation budget re-expanding. +// +// Unlike the other two rotation collectors, this one does not use a +// wrapping cursor: it always selects the current best candidates rather +// than guaranteeing fair coverage of a population, since re-selecting the +// same still-growing parent every pass is exactly the desired behavior, +// not something a fairness guarantee needs to correct for. +std::vector +ReferenceChainTracker::collectLeakAccumulationCandidatesForRotation( + int max_count) { + std::vector selected; + if (max_count <= 0 || _leak_signature_totals.empty()) { + return selected; + } + + // Tier 1: rank signatures by growth since the last pass's snapshot. A + // signature with no prior snapshot (brand new this pass) compares against + // an implicit prev_total of 0 - see _leak_signature_prev_totals' own + // comment for why that is the correct behavior, not a special case. + u64 winning_key = 0; + bool have_winner = false; + u32 best_delta = 0; + for (const auto &kv : _leak_signature_totals) { + u32 prev = 0; + auto prev_it = _leak_signature_prev_totals.find(kv.first); + if (prev_it != _leak_signature_prev_totals.end()) { + prev = prev_it->second; + } + u32 delta = kv.second > prev ? kv.second - prev : 0; + if (delta > 0 && (!have_winner || delta > best_delta)) { + have_winner = true; + best_delta = delta; + winning_key = kv.first; + } + } + // Roll the snapshot forward for the NEXT pass's comparison regardless of + // whether this pass found a winner - a signature that didn't grow this + // pass still needs its current total remembered so a future pass's delta + // is computed against the right baseline, not against however many + // passes ago it was last checked. + _leak_signature_prev_totals = _leak_signature_totals; + if (!have_winner) { + // Nothing grew since last pass - nothing to prioritize this tier this + // time (collectStaleExpandedEntriesForRotation()'s unprioritized + // fallback still covers this population eventually). + return selected; + } + + // Tier 2: within the winning signature only, rank concrete parent objects + // by their own fanout - collected first, then partially sorted, since + // _leak_parent_fanout's total size is what bounds this method's cost (not + // table_size), and is expected to be small (see that map's own comment). + // + // Two parent states qualify: + // - EXPANDED: the ranking's original case - the parent's children were + // admitted once and its EXPANDED state is stale (it has since + // accumulated more children that were never observed), so re-expanding + // it admits the new ones. + // - FRONTIER: the parent is a known holder of watched-klass children + // that has never been expanded at all. Measured live on hotdog (round + // 4, ev-leaktag-onpod-round4): with a 126,895-entry _pending_expand + // backlog draining at ~120-200 objects/min, the growing holders (an + // elementData-sized Object[] of the leaking collection) stay FRONTIER + // for hours, and an EXPANDED-only filter made this tier select ZERO + // every pass while holding 51k known parent candidates + // (leak_accumulation_tags=0 on all 183 passes) - the targeted tier + // going dead in exactly the regime it exists for. Allowing FRONTIER + // parents means a first expansion, and it must JUMP the backlog + // rather than join it: selections are pushed to the FRONT of + // _priority_expand so the next expandFrontier() batch reaches them + // (that deque already held ~1016 stale re-walk entries on the same + // pod; push_back would queue the targeted holder behind all of them). + // + // A FRONTIER-state selection may still have a stale copy sitting in + // _pending_expand (queued there at admission time); the pending-lane + // drain eventually pops that copy and re-expands an already-EXPANDED + // object - one wasted expansion, deduped to edges=0 by the + // ALREADY_ADMITTED check. Bounded (once per selection) and harmless next + // to the value of reaching the holder at all. + std::vector> candidates; // (parent_tag, fanout) + for (const auto &kv : _leak_parent_fanout) { + if (kv.second.signature_key == winning_key && !isQueuedForRotation(kv.first)) { + FrontierEntry entry{}; + if (_frontier->lookup(kv.first, &entry) && + (entry.state == FrontierEntryState::EXPANDED || + entry.state == FrontierEntryState::FRONTIER)) { + candidates.emplace_back(kv.first, kv.second.fanout); + } + } + } + std::sort(candidates.begin(), candidates.end(), + [](const std::pair &a, const std::pair &b) { + return a.second > b.second; + }); + for (const auto &c : candidates) { + if ((int)selected.size() >= max_count || + _priority_expand.size() >= PRIORITY_EXPAND_CAP) { + break; + } + selected.push_back(c.first); + _priority_expand_set.insert(c.first); + FrontierEntry state_entry{}; + bool is_expanded = _frontier->lookup(c.first, &state_entry) && + state_entry.state == FrontierEntryState::EXPANDED; + TEST_LOG("ReferenceChainTracker::" + "collectLeakAccumulationCandidatesForRotation selected " + "parent_tag=%lld state=%s fanout=%u", + (long long)c.first, is_expanded ? "EXPANDED" : "FRONTIER", + c.second); + } + // Place the whole selection at the head of the priority lane, keeping + // the fanout ranking order (see the FRONTIER-state case in the Tier 2 + // comment above for why the head and not the tail): push_front reverses, + // so insert back-to-front. + for (auto it = selected.rbegin(); it != selected.rend(); ++it) { + _priority_expand.push_front(*it); + } + return selected; +} + +// --------------------------------------------------------------------------- +// Retention-edge labels: naming the field each chain hop is retained +// through, from the JVMTI-specification field ordinal captured at +// admission (see FrontierEntry::referrer_field_index's own comment). +// --------------------------------------------------------------------------- +namespace { + +// Fallback label for a hop whose edge is not a field reference (or whose +// field ordinal could not be decoded) - the edge KIND, never a fabricated +// name. Numbering per jvmti.h's jvmtiHeapReferenceKind. +const char *hopEdgeKindLabel(u8 kind) { + switch (kind) { + case JVMTI_HEAP_REFERENCE_CLASS: + return "class"; + case JVMTI_HEAP_REFERENCE_FIELD: + return "field"; + case JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT: + return "element"; + case JVMTI_HEAP_REFERENCE_CLASS_LOADER: + return "class_loader"; + case JVMTI_HEAP_REFERENCE_SIGNERS: + return "signers"; + case JVMTI_HEAP_REFERENCE_PROTECTION_DOMAIN: + return "protection_domain"; + case JVMTI_HEAP_REFERENCE_INTERFACE: + return "interface"; + case JVMTI_HEAP_REFERENCE_STATIC_FIELD: + return "static_field"; + case JVMTI_HEAP_REFERENCE_CONSTANT_POOL: + return "constant_pool"; + case JVMTI_HEAP_REFERENCE_SUPERCLASS: + return "superclass"; + case JVMTI_HEAP_REFERENCE_JNI_GLOBAL: + return "jni_global"; + case JVMTI_HEAP_REFERENCE_SYSTEM_CLASS: + return "system_class"; + case JVMTI_HEAP_REFERENCE_MONITOR: + return "monitor"; + case JVMTI_HEAP_REFERENCE_STACK_LOCAL: + return "stack_local"; + case JVMTI_HEAP_REFERENCE_JNI_LOCAL: + return "jni_local"; + case JVMTI_HEAP_REFERENCE_THREAD: + return "thread"; + case JVMTI_HEAP_REFERENCE_OTHER: + return "other"; + default: + return "unknown"; + } +} + +// Appends the own-declared field names of `cls` to *out, in GetClassFields() +// order. Does NOT delete `cls`'s local ref - the caller owns every jclass +// local ref it passes in and deletes them on all paths exactly once. +// Returns false on any JVMTI failure (caller treats the whole class +// as undecodable - partial lists would silently MISNAME ordinals). +bool appendClassFieldNames(jvmtiEnv *jvmti, JNIEnv *jni, jclass cls, + std::vector *out) { + jint count = 0; + jfieldID *fields = nullptr; + if (jvmti->GetClassFields(cls, &count, &fields) != JVMTI_ERROR_NONE) { + return false; + } + bool ok = true; + for (jint i = 0; i < count; i++) { + char *name = nullptr; + if (jvmti->GetFieldName(cls, fields[i], &name, nullptr, nullptr) != + JVMTI_ERROR_NONE || + name == nullptr) { + ok = false; + break; + } + out->emplace_back(name); + jvmti->Deallocate((unsigned char *)name); + } + jvmti->Deallocate((unsigned char *)fields); + return ok; +} + +// Sums the own-declared field counts of every interface transitively +// implemented/extended by `cls`, each counted once (a diamond interface graph +// shared by two branches must not double-count - the spec's ordinal space +// counts interface fields once each, and HotSpot's transitive_interfaces +// array contains each interface exactly once). `seen` dedupes by raw class +// tag. Returns -1 on JVMTI failure. +// Recursion depth cap: termination is already guaranteed by the `seen` +// tag-dedupe set, but recursion DEPTH equals the number of distinct +// reachable interfaces - a generated/proxied class graph with thousands of +// interfaces would drive equally deep native recursion on the BFS thread +// (each frame also holding a GetImplementedInterfaces result array +// mid-flight). Past the cap the decode fails (same -1 contract as any +// JVMTI error), capping the stack at the caller's frame plus 64. +constexpr int INTERFACE_FIELD_COUNT_MAX_DEPTH = 64; + +// Allocation sanity bound for hopLabelClassFor()'s names vector (sum of +// JVMTI-reported field counts over the transitive interface graph). 8192 +// entries x sizeof(std::string) ≈ 256KB worst case - far above any real +// interface graph, far below a corrupted count turning into a +// gigabyte-scale allocation on the BFS thread. +constexpr jlong MAX_HOP_LABEL_FIELD_BASE = 8192; + +jlong interfaceFieldCount(jvmtiEnv *jvmti, JNIEnv *jni, jclass cls, + std::unordered_set *seen, int depth = 0) { + if (depth > INTERFACE_FIELD_COUNT_MAX_DEPTH) { + return -1; + } + // GetTag is safe on any jclass; the shared class-tag allocator + // (classTagAllocator.h) tags every class resolveLoadedClasses() sees, and + // classes here always come from that set. + jlong tag = 0; + if (jvmti->GetTag(cls, &tag) != JVMTI_ERROR_NONE || tag == 0) { + // Untagged interface: cannot dedupe reliably - fail the whole decode + // rather than risk double counting. + return -1; + } + if (!seen->insert(tag).second) { + return 0; // already counted this interface (a shared subinterface) + } + jint iface_count = 0; + jclass *ifaces = nullptr; + if (jvmti->GetImplementedInterfaces(cls, &iface_count, &ifaces) != + JVMTI_ERROR_NONE) { + return -1; + } + jlong total = 0; + for (jint i = 0; i < iface_count; i++) { + if (ifaces[i] == nullptr) { + continue; + } + // An interface's own fields count toward any implementor's ordinal + // base - matching the spec's "count of the fields in all the interfaces + // implemented by C" (jvmtiHeapReferenceInfoField). + jint field_count = 0; + jfieldID *fields = nullptr; + if (jvmti->GetClassFields(ifaces[i], &field_count, &fields) == + JVMTI_ERROR_NONE) { + total += field_count; + jvmti->Deallocate((unsigned char *)fields); + } + total += interfaceFieldCount(jvmti, jni, ifaces[i], seen, depth + 1); + if (total < 0) { + return -1; + } + jni->DeleteLocalRef(ifaces[i]); + } + jvmti->Deallocate((unsigned char *)ifaces); + return total; +} + +} // namespace + +const ReferenceChainTracker::HopLabelClass * +ReferenceChainTracker::hopLabelClassFor(jvmtiEnv *jvmti, JNIEnv *jni, + jlong class_tag) { + auto it = _hop_label_cache.find(class_tag); + if (it != _hop_label_cache.end()) { + return &it->second; + } + // Bounded: chains reference few distinct referrer classes; a wholesale + // clear at the cap (rather than LRU eviction) keeps this O(1) and is + // correct because the cache is purely derived state - any cleared entry + // is transparently rebuilt on its next hop. + if (_hop_label_cache.size() >= HOP_LABEL_CLASS_CACHE_CAP) { + _hop_label_cache.clear(); + } + HopLabelClass entry{}; + entry.class_tag = class_tag; + entry.decode_failed = true; // until proven otherwise + do { + // GetSuperclass is a JNI (not JVMTI) function - modern JVMTI dropped it + // (the spec delivers superclass references via heap callbacks, + // jvmti.xml's JVMTI_HEAP_REFERENCE_SUPERCLASS note); the rest are JVMTI + // slots. Partial function tables (gtest mock environments stub only + // the slots their tests drive) degrade to kind labels rather than + // calling a null slot. + if (jvmti->functions->GetObjectsWithTags == nullptr || + jvmti->functions->IsInterface == nullptr || + jvmti->functions->GetImplementedInterfaces == nullptr || + jvmti->functions->GetClassFields == nullptr || + jvmti->functions->GetFieldName == nullptr || + jni->functions->GetSuperclass == nullptr) { + break; + } + // Resolve the class object from its raw tag (negative - the shared + // allocator's class tags; GetObjectsWithTags accepts any tag value). + jint count = 0; + jobject *objects = nullptr; + jlong *tags = nullptr; + if (jvmti->GetObjectsWithTags(1, &class_tag, &count, &objects, &tags) != + JVMTI_ERROR_NONE || + count != 1 || objects == nullptr || objects[0] == nullptr) { + // Both result arrays are JVMTI-allocated on success and must be + // Deallocate()d by the caller - same contract as every other + // GetObjectsWithTags() call site in this file. On the error path they + // may or may not have been allocated, hence the null guards. + if (objects != nullptr) { + jvmti->Deallocate((unsigned char *)objects); + } + if (tags != nullptr) { + jvmti->Deallocate((unsigned char *)tags); + } + break; + } + jclass cls = (jclass)objects[0]; + // The arrays were only needed to obtain the class object - the jobject + // handle stays valid on its own - so release them before the (multiple, + // break-exited) decode branches below, which otherwise all leak them. + jvmti->Deallocate((unsigned char *)objects); + jvmti->Deallocate((unsigned char *)tags); + jboolean is_interface = JNI_FALSE; + std::vector names; + bool ok = false; + if (jvmti->IsInterface(cls, &is_interface) == JVMTI_ERROR_NONE) { + if (is_interface) { + // The spec's INTERFACE branch: base = fields of all superinterfaces + // of I, then I's own fields (jvmtiHeapReferenceInfoField). + std::unordered_set seen; + jlong base = interfaceFieldCount(jvmti, jni, cls, &seen); + if (base >= 0) { + // Sanity-bound the allocation before resizing: `base` is a sum of + // JVMTI-reported field counts over the whole transitive interface + // graph, and a corrupted/partial-mock JVMTI table (the code's own + // gtest notes: mock tables report bogus field_count) can report a + // number that turns this into an enormous allocation on the BFS + // thread. Real interface graphs are orders of magnitude below + // this; a decode failure (same as any JVMTI error) is the correct + // outcome for a count that large. + if (base > MAX_HOP_LABEL_FIELD_BASE) { + TEST_LOG("ReferenceChainTracker::hopLabelClassFor interface " + "field base %lld exceeds cap - failing decode", + (long long)base); + if (cls != nullptr) { + jni->DeleteLocalRef(cls); + } + break; + } + names.resize((size_t)base); // positioned but unnamed: ordinal [0, + // base) is interface fields, only + // reachable through an interface + // branch decode of a superinterface + ok = appendClassFieldNames(jvmti, jni, cls, &names); + } + } else { + // The spec's CLASS branch: base = fields of all interfaces + // implemented by C, then the superclass chain root-first + // (java.lang.Object's fields first, C's own last), each class's + // fields in GetClassFields() order. + std::unordered_set seen; + jlong base = interfaceFieldCount(jvmti, jni, cls, &seen); + if (base >= 0) { + // Same allocation sanity bound as the interface branch above. + if (base > MAX_HOP_LABEL_FIELD_BASE) { + TEST_LOG("ReferenceChainTracker::hopLabelClassFor interface " + "field base %lld exceeds cap - failing decode", + (long long)base); + if (cls != nullptr) { + jni->DeleteLocalRef(cls); + } + break; + } + names.resize((size_t)base); + // GetSuperclass walks UP, so gather then append in reverse + // (root first). supers[] holds local refs of every class along + // the way, cls included - deleted together below, exactly once. + jclass supers[128]; + int depth = 0; + jclass k = cls; + while (k != nullptr && + depth < (int)(sizeof(supers) / sizeof(supers[0]))) { + supers[depth++] = k; + k = jni->GetSuperclass(k); + } + ok = (k == nullptr); // deeper than 128 classes: fail rather than + // misname + for (int i = depth - 1; ok && i >= 0; i--) { + ok = appendClassFieldNames(jvmti, jni, supers[i], &names); + } + for (int i = 0; i < depth; i++) { + jni->DeleteLocalRef(supers[i]); + } + // supers[0] IS cls - the loop above already deleted it. Null it so + // the shared cleanup below does not delete the same local ref a + // second time (checked JNI reports an invalid local ref and aborts). + cls = nullptr; + } + } + } + if (cls != nullptr) { + jni->DeleteLocalRef(cls); + } + if (!ok) { + break; + } + entry.field_names = std::move(names); + entry.decode_failed = false; + } while (false); + auto inserted = _hop_label_cache.emplace(class_tag, std::move(entry)); + return &inserted.first->second; +} + +void ReferenceChainTracker::resolveHopEdgeLabel(jvmtiEnv *jvmti, JNIEnv *jni, + ChainHopEdge edge, char *out, + size_t out_cap) { + if (jvmti != nullptr && jni != nullptr && + (edge.edge_kind == JVMTI_HEAP_REFERENCE_FIELD || + edge.edge_kind == JVMTI_HEAP_REFERENCE_STATIC_FIELD) && + edge.field_index >= 0 && edge.referrer_class_tag != 0 && out_cap > 0) { + const HopLabelClass *labels = + hopLabelClassFor(jvmti, jni, edge.referrer_class_tag); + if (labels != nullptr && !labels->decode_failed && + (size_t)edge.field_index < labels->field_names.size()) { + const std::string &name = + labels->field_names[(size_t)edge.field_index]; + if (!name.empty()) { + size_t n = name.size() < out_cap - 1 ? name.size() : out_cap - 1; + memcpy(out, name.data(), n); + out[n] = '\0'; + return; + } + // Empty positioned slot (an interface-field ordinal below the + // class's own base) - fall through to the kind label. + } + } + if (out_cap > 0) { + snprintf(out, out_cap, "%s", hopEdgeKindLabel(edge.edge_kind)); + } +} + +void ReferenceChainTracker::fillHopEdgeLabels( + jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &edges, + std::vector *out) { + char label[MAX_HOP_EDGE_LABEL + 1]; + for (size_t i = 0; i < edges.size() && i < out->size(); i++) { + resolveHopEdgeLabel(jvmti, jni, edges[i], label, sizeof(label)); + (*out)[i].edge_label = label; + } +} + +// --------------------------------------------------------------------------- +// Candidate-scoped reach: bounded descend walks from anchor objects (see +// descendFromAnchor()'s declaration comment, referenceChains.h). Reaching +// the tagged leak instances is a correctness requirement, not a +// throughput optimization: breadth-first FIFO expansion over a rising +// heap can never drain (pod rounds 5-6), so the tagged instances sit +// under holders the ordinary crawl reaches only after hours. +// --------------------------------------------------------------------------- +namespace { + +// Class tags to neither admit nor descend into during a descend walk (see +// PassContext::_no_descend_class_tags' own comment). All three are +// bootstrap-resolvable, so FindClass works from any JNI context; a class +// not yet tagged by resolveLoadedClasses() (tag 0) is simply not added - +// the gate then stays off for that class until a later pass re-resolves. +const char *const kNoDescendClassNames[] = { + "java/lang/ClassLoader", "java/lang/ThreadGroup", + "java/security/ProtectionDomain", +}; + +int resolveNoDescendClassTags(jvmtiEnv *jvmti, JNIEnv *jni, + jlong *out, int cap) { + int count = 0; + for (const char *name : kNoDescendClassNames) { + if (count >= cap) { + break; + } + jclass cls = jni->FindClass(name); + if (cls == nullptr) { + // Not loadable in this JVM (e.g. java.security classes stripped by a + // minimal runtime) - skip; the gate simply does not cover it. + jni->ExceptionClear(); + continue; + } + jlong tag = 0; + if (jvmti->GetTag(cls, &tag) == JVMTI_ERROR_NONE && tag != 0) { + out[count++] = tag; + } + jni->DeleteLocalRef(cls); + } + return count; +} + +// java.lang.ThreadLocal$ThreadLocalMap's class tag for +// walkCandidateThreadLocals()'s anchor gate (see PassContext:: +// _anchor_descend_class_tag's own comment): the value type of BOTH of +// Thread's threadLocals and inheritableThreadLocals fields, and its exact +// class tag is what the anchor gate compares against. The class is +// package-private but already loaded in any JVM that has ever touched a +// ThreadLocal (FindClass resolves by name regardless of access), and the +// class name is stable across JDK 8-26. Returns 0 if not resolvable (not +// yet loaded / FindClass refused) - the caller then walks the Thread's +// edges generically, gated only by the no-descend class set + hop cap, +// rather than skipping the walk entirely. +jlong resolveThreadLocalMapClassTag(jvmtiEnv *jvmti, JNIEnv *jni) { + jlong tag = 0; + jclass cls = jni->FindClass("java/lang/ThreadLocal$ThreadLocalMap"); + if (cls == nullptr) { + jni->ExceptionClear(); + return 0; + } + jvmti->GetTag(cls, &tag); + jni->DeleteLocalRef(cls); + return tag; +} + +} // namespace + +void ReferenceChainTracker::descendFromAnchor( + jvmtiEnv *jvmti, JNIEnv *jni, jobject anchor, jlong anchor_tag, + u32 anchor_depth, jlong anchor_descend_class_tag, int budget, + int *edges_admitted, bool *truncated, bool *frontier_cap_hit, + u64 *safepoint_ticks) { + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + // Bound admission to DESCENT_HOPS below the anchor, still subject to the + // global hop cap. Caveat (accepted, bounded): a pre-existing frontier + // entry reachable inside the subgraph carries its GLOBAL depth (from + // whatever path first admitted it), so the walk can descend more than + // DESCENT_HOPS below the anchor through such an entry - never past + // _hop_cap + the pass deadline though, the same bounds every other walk + // phase lives under. + int descent_cap = (int)anchor_depth + DESCENT_HOPS; + ctx.hop_cap = descent_cap < _hop_cap ? descent_cap : _hop_cap; + ctx.budget = budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + + ctx._no_descend_class_tag_count = + resolveNoDescendClassTags(jvmti, jni, ctx._no_descend_class_tags, + PassContext::NO_DESCEND_CLASS_CAP); + if (anchor_descend_class_tag != 0) { + ctx._descent_anchor_tag = anchor_tag; + ctx._anchor_descend_class_tag = anchor_descend_class_tag; + } + + jvmtiHeapCallbacks callbacks; + memset(&callbacks, 0, sizeof(callbacks)); + callbacks.heap_reference_callback = heapReferenceCallback; + u64 follow_start_ticks = TSC::ticks(); + jvmti->FollowReferences(0, nullptr, anchor, &callbacks, &ctx); + *safepoint_ticks += TSC::ticks() - follow_start_ticks; + *edges_admitted += ctx.edges_admitted; + *truncated = *truncated || ctx.truncated; + *frontier_cap_hit = *frontier_cap_hit || ctx.frontier_cap_hit; +} + +std::vector +ReferenceChainTracker::collectStaticFieldAnchorsForRotation(int max_count) { + std::vector selected; + if (max_count <= 0 || _static_anchor_index.empty()) { + return selected; + } + // Tiered selection over _static_anchor_index (O(anchors) per pass, + // under ONE shared lock - the lookups below are lookupLocked()). The + // tiers, in walk order: + // 0. leak-tagged anchors (entry.leak_tag != 0) - always selected + // first (rare; no cursor needed). These are the only anchors that + // already lead to known-leaking objects. + // 1. FRESH anchors (the _static_anchor_fresh_queue drain) - anchors + // admitted since their last first-look attempt, container-shaped + // OR not-yet-classified. Admission order = sweep order = loaded- + // class order, so a leak holder held by a late-loaded class (the + // hotdog LEAK_BUFFER wrapper: holder class at sweep index 24627 of + // 33270, round-15 measurement) is admitted at the index TAIL - + // the very END of the fair container tier's lap. The measured + // hotdog search lifetime (44-75 passes) is shorter than + // ceil(container_cohort/budget) (1633/16 = 102), so fair-only + // coverage deterministically never reaches it; the fresh lane + // walks it within a pass or two of admission instead. Each anchor + // gets exactly ONE first look: a drain that outranks it (budget + // exhausted by earlier fresh picks) or classifies it as a + // non-container drops it from the queue, and it falls back to the + // fair tiers at its index position - covered eventually, just not + // urgently. Not-yet-classified anchors ride the lane because the + // wrapper admits one pass before reconcileAnchorClassShapes() can + // classify its class; "not-yet-classified" is bounded in practice + // (the shape cache is JVM-lifetime, so only genuinely new classes + // arrive unclassified, and churn classes are lambdas with no + // static fields). + // 2. container-shaped anchors, cursor-fair. A leak holder is + // typically a container, and the container cohort is far smaller + // than the full anchor population (round-14 hotdog measurement: + // 1633-1680 containers of 27739-30711 anchors). Fair so a cohort + // larger than the budget still rotates to coverage instead of + // hammering the same prefix. + // 3. everything else (String/Class/boxed/enum statics), cursor-fair, + // eventually covered within ceil(tier/budget) passes - explicitly + // NOT guaranteed within one search lifetime; that is the accepted + // cost of prioritizing containers (the round-13 starvation + // analysis). + // Eligibility filters (unchanged from the single-cursor version): + // liveness (entry cleared/ABANDONED since indexing), root-attached + // durable root kinds, FRONTIER/EXPANDED states, !isQueuedForRotation. + size_t idx_size = _static_anchor_index.size(); + if (_anchor_container_cursor >= idx_size) { + _anchor_container_cursor = 0; + } + if (_anchor_other_cursor >= idx_size) { + _anchor_other_cursor = 0; + } + struct TierPick { + size_t pos; + jlong tag; + }; + std::vector leak_picks; + std::vector fresh_picks; + std::vector container_picks; + std::vector other_picks; + leak_picks.reserve(16); + // Fresh picks kept by the queue drain (bounded by max_count) - used to + // keep the fair-tier consumption below from double-selecting them. + std::unordered_set fresh_kept_tags; + const size_t fresh_queue_len = _static_anchor_fresh_queue.size(); + _frontier->withSharedLock([&](const FrontierTable *frontier) { + // Index scan: partition every eligible anchor into the leak tier or + // one of the two fair tiers (the fresh lane is decided by the queue + // drain below - a fresh-kept anchor also lands in a fair pick vector + // here and is skipped at consumption time via fresh_kept_tags). + for (size_t i = 0; i < idx_size; i++) { + jlong tag = _static_anchor_index[i]; + FrontierEntry entry{}; + if (!frontier->lookupLocked(tag, &entry) || + entry.parent_tag != 0 || + (entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && + entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) || + (entry.state != FrontierEntryState::FRONTIER && + entry.state != FrontierEntryState::EXPANDED) || + isQueuedForRotation(tag)) { + continue; + } + if (entry.leak_tag != 0) { + leak_picks.push_back(TierPick{i, tag}); + } else if (i < _static_anchor_own_class_tags.size()) { + auto shape_it = + _class_shape_cache.find(_static_anchor_own_class_tags[i]); + if (shape_it != _class_shape_cache.end() && + shape_it->second == (u8)AnchorClassShape::CONTAINER) { + container_picks.push_back(TierPick{i, tag}); + } else { + other_picks.push_back(TierPick{i, tag}); + } + } else { + other_picks.push_back(TierPick{i, tag}); + } + } + // Fresh-lane drain. Every queue entry is popped (its ONE first look + // is spent either way): kept if eligible AND (container-shaped OR + // not-yet-classified) AND room remains in the budget; dropped + // otherwise. Leak-tagged anchors are dropped here - the leak tier + // above already owns them and its picks lead the selection anyway. + // The queue is a contiguous, ordered slice of the index (the two + // append together in addToStaticAnchorIndex; every removal - a cap + // drop, a spent first look - pops from the front), so entries drain + // in lockstep with index positions counting up from + // idx_size - fresh_queue_len: O(1) per entry, no tag search needed. + int fresh_room = max_count - (int)leak_picks.size(); + size_t drain_pos = fresh_queue_len <= idx_size ? idx_size - fresh_queue_len : 0; + while (!_static_anchor_fresh_queue.empty()) { + if (fresh_room <= 0) { + // Budget exhausted before the queue drained: everything + // remaining spends its first look now and falls back to the + // fair tiers at its index position (covered, not urgent). + _static_anchor_fresh_queue.clear(); + break; + } + jlong tag = _static_anchor_fresh_queue.front(); + _static_anchor_fresh_queue.pop_front(); + size_t pos = drain_pos; + drain_pos++; + if (pos >= idx_size || _static_anchor_index[pos] != tag) { + // The suffix-window invariant broke (cannot happen today; + // defensive): fall back to a search rather than mis-shape the + // entry - the queue is small, this is not a hot path once + // healthy. + auto it = + std::find(_static_anchor_index.begin(), + _static_anchor_index.end(), tag); + if (it == _static_anchor_index.end()) { + continue; + } + pos = (size_t)(it - _static_anchor_index.begin()); + } + FrontierEntry entry{}; + if (!frontier->lookupLocked(tag, &entry) || + entry.parent_tag != 0 || + (entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && + entry.root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) || + (entry.state != FrontierEntryState::FRONTIER && + entry.state != FrontierEntryState::EXPANDED) || + isQueuedForRotation(tag) || entry.leak_tag != 0) { + continue; // dead/demoted/queued/leak-tier: first look spent, not + // fresh-kept (the leak tier selects it via the scan if + // it is leak-tagged) + } + bool keep = false; // container or not-yet-classified rides the + // lane; the wrapper admits one pass before + // reconcile can classify its class + if (pos < _static_anchor_own_class_tags.size()) { + auto shape_it = + _class_shape_cache.find(_static_anchor_own_class_tags[pos]); + keep = shape_it == _class_shape_cache.end() || + shape_it->second == (u8)AnchorClassShape::CONTAINER; + } else { + keep = true; // no own-class tag recorded - treat as unknown + } + if (!keep) { + continue; // classified non-container: the other tier owns it + } + fresh_picks.push_back(TierPick{pos, tag}); + fresh_kept_tags.insert(tag); + fresh_room--; + } + }); + // Cursor-fair consumption of one tier: scan picks (sorted by pos by + // construction) starting at entries with pos >= cursor, stop at `want` + // OR at the lap end (NO within-call wrap: re-walking anchors this same + // call already covered would waste walk budget - the leftover budget + // flows to the next tier instead, and the cursor resets to 0 so the + // NEXT call starts a fresh lap). Fresh-kept tags are skipped - the + // fresh lane already selected them this call - but the cursor still + // passes their positions (they were covered this call, in effect). + auto consume_tier_fair = [&](const std::vector &picks, + size_t &cursor, int want) { + int took = 0; + if (want <= 0 || picks.empty()) { + return took; + } + size_t consumed_pos = 0; + for (size_t k = 0; k < picks.size() && took < want; k++) { + const TierPick &p = picks[k]; + if (p.pos < cursor) { + continue; + } + if (fresh_kept_tags.count(p.tag) > 0) { + continue; + } + selected.push_back(p.tag); + consumed_pos = p.pos; + took++; + } + if (took > 0) { + cursor = consumed_pos + 1 >= idx_size ? 0 : consumed_pos + 1; + } + return took; + }; + int budget_left = max_count; + for (const TierPick &p : leak_picks) { + if (budget_left <= 0) { + break; + } + selected.push_back(p.tag); + budget_left--; + } + // Fresh lane: queue order (admission order) so a burst larger than the + // budget spends the oldest first looks first and nothing jumps the + // queue; outranked fresh anchors fall back to the fair tiers at their + // positions (the drain already dropped them from the queue). + for (const TierPick &p : fresh_picks) { + if (budget_left <= 0) { + break; + } + selected.push_back(p.tag); + budget_left--; + } + budget_left -= consume_tier_fair(container_picks, _anchor_container_cursor, + budget_left); + // The other tier is the last consumer of the budget - its leftover has no + // further reader, so don't accumulate it back into budget_left (a dead + // store clang scan-build flags). + consume_tier_fair(other_picks, _anchor_other_cursor, budget_left); + return selected; +} + +void ReferenceChainTracker::pushAtRiskStaticAnchor(jlong tag, u32 klass_id) { + if (_static_anchor_fifo_set.contains(tag)) { + return; + } + if (_static_anchor_fifo.size() >= STATIC_ANCHOR_FIFO_CAP) { + // Cap-full drop. With the per-class quota below this is legitimate + // saturation: a full 1024-entry FIFO necessarily holds >= 16 distinct + // under-quota classes (measured pod contrast, round 15: the cap was + // pinned by three classes' floods - klass 1 at 1396 pushes, klass 215 + // at 1063, klass 1733 at 988+ - so the wrapper's own pushes were + // dropped here and B' was dead for exactly the holder it exists for). + return; + } + auto count_it = _static_anchor_fifo_klass_counts.find(klass_id); + if (count_it != _static_anchor_fifo_klass_counts.end() && + count_it->second >= STATIC_ANCHOR_ATRISK_PER_KLASS_CAP) { + // Per-class quota drop: this class already holds its share of the + // lane, and its oldest entry drains within a few passes + // (STATIC_ANCHOR_FIFO_DRAIN=16/pass). A dropped push is retried by + // the feed's next event (the next static edge / next demotion), so + // nothing is lost - the entry just cannot crowd out every other + // class's repair. + return; + } + if (count_it == _static_anchor_fifo_klass_counts.end()) { + count_it = _static_anchor_fifo_klass_counts.emplace(klass_id, 0U).first; + } + count_it->second++; + _static_anchor_fifo.push_back(AtRiskAnchor{tag, klass_id}); + _static_anchor_fifo_set.insert(tag); +} + +void ReferenceChainTracker::addToStaticAnchorIndex(jlong tag, + jlong own_class_tag, + u8 root_kind) { + if (root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD && + root_kind != (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL) { + return; + } + // Dedup: a push at first admission and another at upgrade would double-add. + // The round-13 pod measurement sized the real anchor population at ~28k + // per search - a linear scan per add is O(n^2) over a search (observed + // cost ~2ms/pass amortized, wasted inside the sweep callback), so dedupe + // via a companion hash set (the PriorityExpandSet pattern, but unbounded: + // the set is cleared with the index on restart). + if (!_static_anchor_index_tags.insert(tag).second) { + return; + } + _static_anchor_index.push_back(tag); + _static_anchor_own_class_tags.push_back(own_class_tag); + // Fresh lane (round 15, _static_anchor_fresh_queue's own comment): a + // newly admitted anchor gets ONE first-look walk priority ahead of the + // fair cursors. Cap-drop from the front: the oldest fresh chances fall + // back to the fair tiers (their index positions are unchanged) - never + // lost, only deprioritized, same as a drain that outranks them. + _static_anchor_fresh_queue.push_back(tag); + if (_static_anchor_fresh_queue.size() > STATIC_ANCHOR_FRESH_CAP) { + _static_anchor_fresh_queue.pop_front(); + } +} + +bool ReferenceChainTracker::resolveContainerInterfaceTags( + jvmtiEnv *jvmti, JNIEnv *jni) { + if (_collection_iface_class_tag != 0 && _map_iface_class_tag != 0) { + return true; + } + // resolveLoadedClasses() tags every loaded class (including these + // bootstrap interfaces) with its class-tag-allocator tag (a NEGATIVE + // value - see nextClassTag()'s own comment) before any + // anchor can be admitted, but classify defensively: if an interface + // object somehow carries no tag yet, mint one via the shared allocator + // (same sequence resolveLoadedClasses() itself uses) so the comparison + // below is well-defined. + struct Iface { + const char *name; + jlong *tag_out; + }; + Iface ifaces[2] = {{"java/util/Collection", &_collection_iface_class_tag}, + {"java/util/Map", &_map_iface_class_tag}}; + for (const Iface &iface : ifaces) { + if (*iface.tag_out != 0) { + continue; + } + jclass local = jni->FindClass(iface.name); + if (jniExceptionCheck(jni) || local == nullptr) { + jni->ExceptionClear(); + return false; + } + jlong tag = 0; + bool ok = jvmti->GetTag(local, &tag) == JVMTI_ERROR_NONE; + if (ok && tag == 0) { + tag = nextClassTag(); + ok = jvmti->SetTag(local, tag) == JVMTI_ERROR_NONE; + } + if (ok && tag != 0) { + *iface.tag_out = tag; + } + jni->DeleteLocalRef(local); + if (!ok) { + return false; + } + } + return _collection_iface_class_tag != 0 && _map_iface_class_tag != 0; +} + +bool ReferenceChainTracker::classImplementsContainerOrMap(jvmtiEnv *jvmti, + JNIEnv *jni, + jclass klass) { + // BFS over the superclass chain + every visited class's interfaces, + // comparing GetTag() against the two cached interface class tags. + // Interface diamonds exist (e.g. both List and Set through Collection), + // so a visited set (by class tag) is required for termination; the + // visited set doubles as the memo the caller caches per class tag. + std::vector work; + std::unordered_set visited; + work.push_back(klass); + bool found = false; + int hops = 0; + while (!found && !work.empty() && hops++ < 64) { + jclass cur = work.back(); + work.pop_back(); + // Delete-on-all-exits discipline: `cur` is a minted local (unless it is + // the caller-provided klass) popped from `work` - every exit from this + // iteration must delete it exactly once, including the two early + // continues below (an earlier draft leaked the local ref there, pinning + // the class against unload for the process lifetime on the never- + // detaching BFS thread). + jlong cur_tag = 0; + if (jvmti->GetTag(cur, &cur_tag) != JVMTI_ERROR_NONE || cur_tag == 0) { + if (cur != klass) { + jni->DeleteLocalRef(cur); + } + continue; + } + if (visited.count(cur_tag) > 0) { + if (cur != klass) { + jni->DeleteLocalRef(cur); + } + continue; + } + visited.insert(cur_tag); + if (cur_tag == _collection_iface_class_tag || + cur_tag == _map_iface_class_tag) { + found = true; + // The popped `cur` ref never reaches the loop's bottom delete. + if (cur != klass) { + jni->DeleteLocalRef(cur); + } + break; + } + jclass super = jni->GetSuperclass(cur); + if (!jniExceptionCheck(jni) && super != nullptr) { + work.push_back(super); + } else { + jni->ExceptionClear(); + } + jint iface_count = 0; + jclass *ifaces = nullptr; + if (jvmti->GetImplementedInterfaces(cur, &iface_count, &ifaces) == + JVMTI_ERROR_NONE && + ifaces != nullptr) { + for (jint i = 0; i < iface_count; i++) { + if (ifaces[i] != nullptr) { + work.push_back(ifaces[i]); + } + } + jvmti->Deallocate((unsigned char *)ifaces); + } + // `cur` is either the caller-provided klass (caller-managed ref - NOT + // deleted here) or a ref this walk minted (GetSuperclass/ + // GetImplementedInterfaces locals, deleted immediately after use). + if (cur != klass) { + jni->DeleteLocalRef(cur); + } + } + // Single exit: every remaining ref minted into `work` (early hop-bound + // exit or the found-break) is deleted here rather than leaking locals + // for the process lifetime (the engine thread never detaches). + for (jclass r : work) { + if (r != nullptr && r != klass) { + jni->DeleteLocalRef(r); + } + } + return found; +} + +void ReferenceChainTracker::reconcileAnchorClassShapes(jvmtiEnv *jvmti, + JNIEnv *jni) { + if (jni == nullptr) { + return; + } + if (_static_anchor_own_class_tags.empty()) { + return; + } + // Collect up to ANCHOR_SHAPE_RECONCILE_BUDGET distinct class tags that + // appear in the anchor index but are not yet classified. + std::vector unknown; + unknown.reserve(8); + std::unordered_set seen; + for (jlong class_tag : _static_anchor_own_class_tags) { + if (class_tag == 0 || seen.count(class_tag) > 0 || + _class_shape_cache.count(class_tag) > 0) { + continue; + } + seen.insert(class_tag); + unknown.push_back(class_tag); + if ((int)unknown.size() >= ANCHOR_SHAPE_RECONCILE_BUDGET) { + break; + } + } + if (unknown.empty()) { + return; + } + if (!resolveContainerInterfaceTags(jvmti, jni)) { + return; + } + // The per-class interface walk below mints local refs (GetSuperclass, + // GetImplementedInterfaces) that are only deleted as the BFS pops + // them; bound the outstanding count explicitly rather than relying on + // the JVM to grow the local-ref table. + if (jni->EnsureLocalCapacity(512) < 0 || jniExceptionCheck(jni)) { + jni->ExceptionClear(); + return; + } + // One GetObjectsWithTags call resolves the class objects for the whole + // batch (class objects are tagged with their class tags). + jint obj_count = 0; + jobject *objs = nullptr; + jlong *obj_tags = nullptr; + if (jvmti->GetObjectsWithTags((jint)unknown.size(), unknown.data(), + &obj_count, &objs, &obj_tags) != + JVMTI_ERROR_NONE || + obj_count <= 0) { + if (objs != nullptr) { + jvmti->Deallocate((unsigned char *)objs); + } + if (obj_tags != nullptr) { + jvmti->Deallocate((unsigned char *)obj_tags); + } + return; + } + for (jint i = 0; i < obj_count; i++) { + jclass klass = (jclass)objs[i]; + jlong class_tag = obj_tags[i]; + // class tags are NEGATIVE (a namespace disjoint from positive + // frontier tags); 0 means the object was never tagged - skip only that. + if (class_tag == 0 || klass == nullptr) { + if (klass != nullptr) { + jni->DeleteLocalRef(klass); + } + continue; + } + AnchorClassShape shape = classImplementsContainerOrMap(jvmti, jni, klass) + ? AnchorClassShape::CONTAINER + : AnchorClassShape::NON_CONTAINER; + _class_shape_cache[class_tag] = (u8)shape; + // GetObjectsWithTags() returned a local ref for every resolved class - + // this runs on the long-lived BFS thread, where undeleted locals + // accumulate until detach and pin their classes against unload. + jni->DeleteLocalRef(klass); + } + jvmti->Deallocate((unsigned char *)objs); + jvmti->Deallocate((unsigned char *)obj_tags); +} + +int ReferenceChainTracker::drainStaticAnchorFifo(int max_count, + std::vector &out) { + if (max_count <= 0 || _static_anchor_fifo.empty()) { + return 0; + } + int drained = 0; + while (drained < max_count && !_static_anchor_fifo.empty()) { + AtRiskAnchor entry = _static_anchor_fifo.front(); + _static_anchor_fifo.pop_front(); + auto count_it = _static_anchor_fifo_klass_counts.find(entry.klass_id); + if (count_it != _static_anchor_fifo_klass_counts.end() && + --count_it->second == 0) { + // Erased at zero so the map is bounded by the FIFO's live contents + // (<= 1024 distinct classes), not by the search lifetime. + _static_anchor_fifo_klass_counts.erase(count_it); + } + out.push_back(entry); + drained++; + } + _static_anchor_fifo_set.rebuildFrom(_static_anchor_fifo); + return drained; +} + +void ReferenceChainTracker::requeueStaticAnchorFifoFront( + const std::vector &entries) { + if (entries.empty()) { + return; + } + // Reverse order onto the front preserves the tags' relative FIFO order + // (push_front of the LAST entry first leaves the FIRST entry at the + // deque's front). The tags were popped by this pass's + // drainStaticAnchorFifo() and nothing runs a sweep between that drain and + // here, so no tag can already be in the deque - rebuildFrom() would + // silently keep the FIRST slot for a duplicate, but there are none by + // construction. Re-increment each entry's class occupancy: drain + // decremented it, and the requeued entry occupies a FIFO slot again + // exactly as before the drain. + for (size_t i = entries.size(); i-- > 0;) { + _static_anchor_fifo_klass_counts[entries[i].klass_id]++; + _static_anchor_fifo.push_front(entries[i]); + } + _static_anchor_fifo_set.rebuildFrom(_static_anchor_fifo); +} + +void ReferenceChainTracker::walkStaticFieldAnchors( + jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &anchor_tags, + int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, + u64 *safepoint_ticks, std::vector *unwalked) { + if (anchor_tags.empty()) { + return; + } + // Resolve all anchors with ONE GetObjectsWithTags call - the call's + // O(tag_map) cost is the dominant term on a large tag map (pod round 4: + // 14-41ms floor), exactly why expandFrontier() batches its own resolves. + jint resolved_count = 0; + jobject *objects = nullptr; + jlong *resolved_tags = nullptr; + if (jvmti->GetObjectsWithTags((jint)anchor_tags.size(), anchor_tags.data(), + &resolved_count, &objects, + &resolved_tags) != JVMTI_ERROR_NONE) { + return; + } + int walked = 0; + // First index the walk did NOT consume (breaks before an anchor's walk + // report i, breaks after report i+1; a completed loop keeps the + // resolved_count sentinel). The un-walked set is only meaningful at a + // break - the caller requeues FIFO-sourced anchors, collector-sourced + // ones keep their own cursor retention, and dead tags never resolved + // are intentionally absent (they must not be requeued anywhere). + jint first_unwalked = resolved_count; + // First index whose local ref has not been deleted yet. Every break path + // deletes objects[i] before breaking, so anything at or after i+1 still + // holds a live local ref and must be cleaned up below - this runs on the + // long-lived BFS thread, where undeleted locals accumulate until detach + // and pin their objects against collection. + jint first_undeleted = resolved_count; + for (jint i = 0; i < resolved_count; i++) { + FrontierEntry entry{}; + if (!_frontier->lookup(resolved_tags[i], &entry)) { + // Dead-or-stale between selection and here - skip; release machinery + // owns dead-entry cleanup, never here. + jni->DeleteLocalRef(objects[i]); + continue; + } + int remaining = budget - *edges_admitted; + if (remaining <= 0) { + jni->DeleteLocalRef(objects[i]); + first_unwalked = i; + first_undeleted = i + 1; + break; + } + int edges_before = *edges_admitted; + descendFromAnchor(jvmti, jni, objects[i], resolved_tags[i], entry.depth, + /*anchor_descend_class_tag=*/0, remaining, edges_admitted, + truncated, frontier_cap_hit, safepoint_ticks); + TEST_LOG("ReferenceChainTracker::walkStaticFieldAnchors anchor walk " + "outcome tag=%lld edges=%d truncated=%d cap_hit=%d", + (long long)resolved_tags[i], *edges_admitted - edges_before, + (int)*truncated, (int)*frontier_cap_hit); + walked++; + jni->DeleteLocalRef(objects[i]); + if (*truncated && !*frontier_cap_hit) { + // Budget/deadline exhausted mid-set - remaining anchors keep their + // rotation turn via the cursor next pass (the wrapping cursor already + // tolerates a short selection). + first_unwalked = i + 1; + first_undeleted = i + 1; + break; + } + if (*frontier_cap_hit) { + first_unwalked = i + 1; + first_undeleted = i + 1; + break; + } + } + // Release the local refs of anchors the early exits above skipped - each + // break only deleted its own objects[i]. + for (jint i = first_undeleted; i < resolved_count; i++) { + jni->DeleteLocalRef(objects[i]); + } + if (unwalked != nullptr && first_unwalked < resolved_count) { + unwalked->insert(unwalked->end(), resolved_tags + first_unwalked, + resolved_tags + resolved_count); + } + jvmti->Deallocate((unsigned char *)objects); + jvmti->Deallocate((unsigned char *)resolved_tags); + TEST_LOG_SUMMARY("ReferenceChainTracker::walkStaticFieldAnchors selected=%zu " + "walked=%d edges_admitted=%d truncated=%d frontier_cap_hit=%d", + anchor_tags.size(), walked, *edges_admitted, (int)*truncated, + (int)*frontier_cap_hit); +} + +void ReferenceChainTracker::walkCandidateThreadLocals( + jvmtiEnv *jvmti, JNIEnv *jni, int budget, int *edges_admitted, + bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks) { + if (_candidate_count <= 0) { + return; + } + // Flatten the per-slot qualifying-tid snapshot into (slot, tid) pairs, + // then walk up to THREAD_WALK_MAX_ANCHORS of them per pass, rotating via + // _thread_walk_anchor_cursor so every qualifying tid gets a turn within + // ceil(total / THREAD_WALK_MAX_ANCHORS) passes instead of always walking + // the first candidates' tids. + // Compile-time bound on the stack footprint: two worst-case arrays sized + // by the product of constants on the BFS thread's frame. If either + // constant is ever raised, this fails the build instead of silently + // multiplying stack usage (current footprint: 2 x 80 x 4B = 640B). + static_assert(MAX_CANDIDATE_QUALIFYING_TIDS * MAX_LEAK_CANDIDATES_FROM_LT * + sizeof(int) * 2 <= + 2048, + "walkCandidateThreadLocals stack arrays must stay small"); + int slot[MAX_CANDIDATE_QUALIFYING_TIDS * MAX_LEAK_CANDIDATES_FROM_LT]; + jint tid[sizeof(slot) / sizeof(slot[0])]; + int total = 0; + for (int s = 0; s < _candidate_count; s++) { + for (int q = 0; q < _candidate_qualifying_tid_count[s]; q++) { + if (total >= (int)(sizeof(slot) / sizeof(slot[0]))) { + break; + } + slot[total] = s; + tid[total] = _candidate_qualifying_tids[s][q]; + total++; + } + } + if (total == 0) { + return; + } + jlong descend_class_tag = resolveThreadLocalMapClassTag(jvmti, jni); + if (_thread_walk_anchor_cursor < 0 || + _thread_walk_anchor_cursor >= total) { + _thread_walk_anchor_cursor = 0; + } + int walked = 0; + const int start = _thread_walk_anchor_cursor; + int i = start; + do { + jobject thread_obj; + { + MutexLocker ml(_thread_objects_lock); + auto it = _thread_objects.find(tid[i]); + if (it == _thread_objects.end()) { + thread_obj = nullptr; // Thread died/never registered - skip + } else { + thread_obj = it->second; + } + } + if (thread_obj != nullptr) { + // Anchor admission, idempotent across passes: a tag that still maps + // to a live entry is reused as-is (the Thread object is commonly + // root-attached by root enumeration already); a stale positive tag + // (search restart reissued tags from 1, releaseSearchTags() did not + // clear this object because release only touches FrontierTable + // entries) must be re-minted, otherwise the walk would parent new + // children onto a dead table slot or, worse, onto the entry a + // reissued tag now belongs to. + jlong anchor_tag = getTag(jvmti, thread_obj); + u32 anchor_depth = 0; + FrontierEntry anchor_entry{}; + if (anchor_tag > 0 && _frontier->lookup(anchor_tag, &anchor_entry)) { + anchor_depth = anchor_entry.depth; + } else { + jclass thread_class = jni->GetObjectClass(thread_obj); + jlong class_tag = 0; + jvmti->GetTag(thread_class, &class_tag); + u32 referrer_klass = classTags()->resolve(class_tag); + jlong fresh_tag = tagObject(jvmti, thread_obj); + if (fresh_tag != 0) { + if (_frontier->insert(fresh_tag, 0, referrer_klass, 0, + FrontierEntryState::FRONTIER, + (u8)JVMTI_HEAP_REFERENCE_THREAD, class_tag)) { + anchor_tag = fresh_tag; + } else { + // Insert failed (frontier cap/capacity): roll the JVMTI tag + // back, exactly like tagAsRootForTest()'s own rollback. Without + // this the Thread object keeps a minted frontier tag no table + // entry records - releaseSearchTags() can never clear it (it + // only iterates tags present in the frontier), so after + // restartSearch() rewinds _next_tag the same numeric tag can + // be reissued to an unrelated object while this Thread still + // carries it. + clearTag(jvmti, thread_obj); + anchor_tag = 0; + } + } else { + anchor_tag = 0; + } + jni->DeleteLocalRef(thread_class); + } + if (anchor_tag != 0) { + int remaining = budget - *edges_admitted; + if (remaining > 0) { + descendFromAnchor(jvmti, jni, thread_obj, anchor_tag, anchor_depth, + descend_class_tag, remaining, edges_admitted, + truncated, frontier_cap_hit, safepoint_ticks); + walked++; + } + } + } + i = (i + 1) % total; + if (*frontier_cap_hit || walked >= THREAD_WALK_MAX_ANCHORS || + budget - *edges_admitted <= 0) { + break; + } + } while (i != start); + _thread_walk_anchor_cursor = i; + TEST_LOG_SUMMARY("ReferenceChainTracker::walkCandidateThreadLocals candidates=%d " + "tids=%d walked=%d edges_admitted=%d truncated=%d " + "frontier_cap_hit=%d", + _candidate_count, total, walked, *edges_admitted, (int)*truncated, + (int)*frontier_cap_hit); +} + +void ReferenceChainTracker::registerExistingThreads(jvmtiEnv *jvmti, + JNIEnv *jni) { + if (!_enabled || jvmti == nullptr || jni == nullptr) { + return; + } + // Register PRE-EXISTING threads into the tid -> Thread-object registry + // (see _thread_objects' own comment): Profiler::onThreadStart() only sees + // threads started after the recording began, and a leaking thread is + // typically alive since well before the profiler attached (observed live: + // a leaking thread that started before the profiler to seed its fixture + // stayed unregistered, walkCandidateThreadLocals() + // reporting walked=0 while the only thing standing between the walk and + // the tagged chunks was the registry lookup). Same native-tid mapping the + // profiler's own thread-name refresh uses for these same pre-existing + // threads (Profiler::updateThreadName -> JVMThread::nativeThreadId, + // profiler.cpp). Best-effort per thread: a thread whose native tid cannot + // be resolved is simply skipped - a later ThreadEnd unregister for it is + // a harmless no-op lookup miss. Runs from Profiler::start() rather than + // this class's own start() so gtest binaries that call start() directly + // with partial mock JVMTI tables (no GetAllThreads slot) never reach the + // real-call path - only the real profiler lifecycle guarantees a fully + // populated JVMTI table here. + jint thread_count = 0; + jthread *thread_objects = nullptr; + if (jvmti->GetAllThreads(&thread_count, &thread_objects) != JVMTI_ERROR_NONE) { + return; + } + for (jint i = 0; i < thread_count; i++) { + jthread thread = thread_objects[i]; + if (thread == nullptr) { + continue; + } + int tid = JVMThread::nativeThreadId(jni, thread); + if (jni->ExceptionCheck()) { + jni->ExceptionClear(); + continue; + } + if (tid >= 0) { + registerThreadObject(jni, tid, thread); + } + jni->DeleteLocalRef(thread); + } + jvmti->Deallocate((unsigned char *)thread_objects); +} + +void ReferenceChainTracker::registerThreadObject(JNIEnv *jni, int tid, + jthread thread) { + if (!_enabled || jni == nullptr || thread == nullptr) { + return; + } + jobject ref = jni->NewGlobalRef(thread); + if (ref == nullptr) { + return; + } + MutexLocker ml(_thread_objects_lock); + auto it = _thread_objects.find(tid); + if (it != _thread_objects.end()) { + // Same deferred-deletion rule as unregisterThreadObject(): a walk may + // still hold a copy of the replaced ref. + _thread_refs_pending_delete.push_back(it->second); + } + _thread_objects[tid] = ref; +} + +void ReferenceChainTracker::unregisterThreadObject(JNIEnv *jni, int tid) { + if (jni == nullptr) { + return; + } + MutexLocker ml(_thread_objects_lock); + auto it = _thread_objects.find(tid); + if (it != _thread_objects.end()) { + // NOT DeleteGlobalRef() here: walkCandidateThreadLocals() may have + // already copied this jobject out of the map (lock released) and still + // be using it as a FollowReferences anchor - deleting a global ref + // invalidates it for every other JNI call (JNI spec), so deletion is + // deferred to releaseEndedThreadRefs() on the BFS thread (see + // _thread_refs_pending_delete's comment). + _thread_refs_pending_delete.push_back(it->second); + _thread_objects.erase(it); + } +} + +void ReferenceChainTracker::releaseEndedThreadRefs(JNIEnv *jni) { + if (jni == nullptr) { + return; + } + std::vector pending; + { + MutexLocker ml(_thread_objects_lock); + pending.swap(_thread_refs_pending_delete); + } + for (size_t i = 0; i < pending.size(); i++) { + jni->DeleteGlobalRef(pending[i]); + } +} + +void ReferenceChainTracker::releaseAllThreadObjects(JNIEnv *jni) { + if (jni == nullptr) { + return; + } + // Recording stop: the BFS thread is joined (Profiler::stop() order), so no + // walk phase can hold a copied ref - the deferred-deletion indirection of + // unregisterThreadObject() is unnecessary here and every ref can go now. + std::vector pending; + { + MutexLocker ml(_thread_objects_lock); + for (auto &kv : _thread_objects) { + pending.push_back(kv.second); + } + _thread_objects.clear(); + pending.insert(pending.end(), _thread_refs_pending_delete.begin(), + _thread_refs_pending_delete.end()); + _thread_refs_pending_delete.clear(); + } + for (size_t i = 0; i < pending.size(); i++) { + jni->DeleteGlobalRef(pending[i]); + } +} + +// --------------------------------------------------------------------------- +// Manual walk driver - IterateOverReachableObjects root/stack-ref enumeration +// plus expandFrontier()'s batched array-holder FollowReferences hop expansion. +// The only path driven by runPass() below. +// --------------------------------------------------------------------------- + +namespace { +// jvmtiHeapRootKind (IterateOverReachableObjects's root/stack-ref callbacks, +// ordinals 1-7) and jvmtiHeapReferenceKind (FrontierEntry::root_kind's own +// type, FollowReferences' callback, ordinals 8/21-27) are different, disjoint +// enums per the real jvmti.h - storing a raw jvmtiHeapRootKind value into +// root_kind unmodified would make flightRecorder.cpp's rootKindName() report +// "unknown" for every root-callback-attributed chain. Every jvmtiHeapRootKind +// value maps onto its jvmtiHeapReferenceKind namesake; there is no root-kind +// equivalent of STATIC_FIELD (that value only ever arises from +// heapReferenceCallback()'s own referrer-is-a-tagged-class case), so it is +// never produced here. +u8 translateHeapRootKind(jvmtiHeapRootKind root_kind) { + switch (root_kind) { + case JVMTI_HEAP_ROOT_JNI_GLOBAL: + return (u8)JVMTI_HEAP_REFERENCE_JNI_GLOBAL; + case JVMTI_HEAP_ROOT_SYSTEM_CLASS: + return (u8)JVMTI_HEAP_REFERENCE_SYSTEM_CLASS; + case JVMTI_HEAP_ROOT_MONITOR: + return (u8)JVMTI_HEAP_REFERENCE_MONITOR; + case JVMTI_HEAP_ROOT_STACK_LOCAL: + return (u8)JVMTI_HEAP_REFERENCE_STACK_LOCAL; + case JVMTI_HEAP_ROOT_JNI_LOCAL: + return (u8)JVMTI_HEAP_REFERENCE_JNI_LOCAL; + case JVMTI_HEAP_ROOT_THREAD: + return (u8)JVMTI_HEAP_REFERENCE_THREAD; + case JVMTI_HEAP_ROOT_OTHER: + default: + return (u8)JVMTI_HEAP_REFERENCE_OTHER; + } +} + +} // namespace + +jvmtiIterationControl JNICALL ReferenceChainTracker::heapRootCallback( + jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, + void *user_data) { + PassContext *ctx = (PassContext *)user_data; + if (ctx->tracker->_abort_pass_requested.load(std::memory_order_relaxed)) { + ctx->truncated = true; + return JVMTI_ITERATION_ABORT; + } + if (ctx->truncated) { + return JVMTI_ITERATION_ABORT; + } + + u32 referrer_klass = ctx->tracker->classTags()->resolve(class_tag); + u8 translated_root_kind = translateHeapRootKind(root_kind); + AdmitResult result = ctx->tracker->admitObject( + ctx->frontier, ctx->hop_cap, ctx->budget, &ctx->edges_admitted, tag_ptr, + /*parent_tag=*/0, referrer_klass, /*depth=*/0, translated_root_kind, + class_tag); + switch (result) { + case AdmitResult::BUDGET_EXHAUSTED: + ctx->truncated = true; + return JVMTI_ITERATION_ABORT; + case AdmitResult::FRONTIER_CAP_HIT: + ctx->truncated = true; + ctx->frontier_cap_hit = true; + return JVMTI_ITERATION_ABORT; + case AdmitResult::ALREADY_ADMITTED: + // Rediscovery via a second heap root - either later in this same pass's + // root enumeration, or in a later pass re-enumerating roots entirely + // (design doc's durability tie-break / "opportunistic upgrade" / "Fix for + // root-attribution staleness" point 1): apply the + // same durability ranking admitObject() would have used on first + // discovery, upgrading root_kind if this root is more durable than + // whatever is currently recorded. Restricted to root-attached entries + // only (parent_tag == 0) - see maybeUpgradeRootAttachedRootKind()'s own + // comment for why. + ctx->tracker->maybeUpgradeRootAttachedRootKind(ctx->frontier, *tag_ptr, + translated_root_kind); + break; + default: + break; + } + return JVMTI_ITERATION_CONTINUE; +} + +jvmtiIterationControl JNICALL ReferenceChainTracker::stackRefCallback( + jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, + jlong thread_tag, jint depth, jmethodID method, jint slot, + void *user_data) { + // Stack-local/JNI-local roots carry thread/frame/slot detail JVMTI reports + // via this callback's richer shape, but FrontierEntry has nowhere to + // record it (depth/method/slot are not part of the record) - admission is + // otherwise identical to heapRootCallback() above, so this just forwards. + return heapRootCallback(root_kind, class_tag, size, tag_ptr, user_data); +} + +void ReferenceChainTracker::runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, + bool run_root_enum, + int root_enum_budget, + int expand_budget, + int *edges_admitted, + bool *truncated, + bool *frontier_cap_hit, + u64 *safepoint_ticks) { + assert(!t_inGCCallback && + "IterateOverReachableObjects/FollowReferences are JVMTI " + "Heap-category calls and must not be made from " + "GarbageCollectionStart/Finish"); + + // Safe point to delete the global refs of threads that ended since the + // last drain: this runs on the BFS thread before any walk phase, and refs + // erased from _thread_objects (unregisterThreadObject()) can no longer be + // copied out by walkCandidateThreadLocals(), so no walk holds them. + releaseEndedThreadRefs(jni); + + // Same safe point for the heap-callback's deferred resolved-chain + // invalidations: applying them here (outside any walk) keeps + // _resolved_chains_lock traffic off the FollowReferences pause entirely. + drainPendingChainInvalidations(); + + *safepoint_ticks = 0; + + // Shared wall-clock ceiling for this whole call's static-field sweep, + // expandFrontier(), and rotation sub-calls below (see _pass_deadline_ns's + // own comment) - deliberately NOT applied to root/stack-ref enumeration + // itself, which is instead cadence-gated by run_root_enum/ + // ROOT_ENUM_MIN_INTERVAL_NS. + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + + *edges_admitted = 0; + *truncated = false; + *frontier_cap_hit = false; + + // Reserve a slice for rotation up front, across all three tiers (see + // ROOT_KIND_ROTATION_BUDGET/LEAK_ACCUMULATION_ROTATION_BUDGET/ + // STALE_EXPANDED_ROTATION_BUDGET's own comments) so rotation still gets to + // run this pass even when ordinary work below spends everything else and + // truncates. Also capped at half of expand_budget: without that cap, a + // pacing-throttled pass (expand_budget down near MIN_EFFECTIVE_BUDGET) + // would hand rotation its full reservation and leave ordinary expansion + // with 0 - exactly the priority inversion this reservation exists to + // avoid, just for the other side. Capping at half means each side + // degrades proportionally as pacing throttles down, instead of either one + // hitting a hard 0. + int rotation_reserved_budget = std::min( + expand_budget / 2, ROOT_KIND_ROTATION_BUDGET + + LEAK_ACCUMULATION_ROTATION_BUDGET + + STALE_EXPANDED_ROTATION_BUDGET); + int budget = expand_budget - rotation_reserved_budget; + + // Root/stack-ref enumeration alone (unlike a root-seeded FollowReferences + // call on the fallback path) never discovers a root's own transitive + // children - IterateOverReachableObjects's root/stack-ref callbacks are + // given no oop, only a tag_ptr (see heapRootCallback()'s own comment) - so + // even when it runs this pass, the expandFrontier() call below is still + // needed to make any further progress. Gated behind run_root_enum (see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment) since the call's fixed + // root-walk-and-dispatch cost is paid in full every time it runs, + // regardless of budget. + if (run_root_enum) { + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx.hop_cap = _hop_cap; + ctx.budget = root_enum_budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + + u64 root_enum_start_ticks = TSC::ticks(); + jvmtiError root_err = jvmti->IterateOverReachableObjects( + heapRootCallback, stackRefCallback, /*object_ref_callback=*/nullptr, + &ctx); + *safepoint_ticks += TSC::ticks() - root_enum_start_ticks; + + // expand_budget is spent independently of root_enum_budget below (see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment) - ctx.edges_admitted is + // written straight into *edges_admitted so the static-field/expand/ + // rotation budget math below is never shrunk by whatever root + // enumeration admitted. + *edges_admitted = ctx.edges_admitted; + _last_root_enum_ns = OS::nanotime(); + + if (root_err != JVMTI_ERROR_NONE) { + *truncated = true; + *frontier_cap_hit = false; + _root_enum_truncated_last_time = false; + return; + } + if (ctx.truncated) { + *truncated = true; + *frontier_cap_hit = ctx.frontier_cap_hit; + // Only a budget-exhausted truncation (not a frontier-cap-hit, which + // abandons the search outright) is grounds to retry root enumeration + // on the very next pass - see _root_enum_truncated_last_time's own + // comment. + _root_enum_truncated_last_time = !ctx.frontier_cap_hit; + return; + } + _root_enum_truncated_last_time = false; + } + + int expand_phase_edges_admitted = 0; + + // Candidate-scoped reach, prong 1: descend-walk the current candidates' + // qualifying threads' ThreadLocalMap subgraphs BEFORE any breadth-first + // work this pass - reaching the tagged instances under a thread-retained + // holder must not queue behind the ordinary backlog (see + // walkCandidateThreadLocals()'s own comment). Own deadline slice + // (per-sub-op reset, same as expand/rotation below) and its own budget + // draw from the ordinary expand slice; a truncation here behaves exactly + // like a truncated static sweep below (search stays RUNNING, remaining + // work resumes next pass). + if (_candidate_count > 0) { + int thread_walk_edges_admitted = 0; + bool thread_walk_truncated = false; + bool thread_walk_frontier_cap_hit = false; + // Give the thread walk its own fresh deadline so the root-enum walk + // above never eats its slice (per-sub-op reset rationale, see expand + // below). + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + walkCandidateThreadLocals(jvmti, jni, budget, &thread_walk_edges_admitted, + &thread_walk_truncated, + &thread_walk_frontier_cap_hit, safepoint_ticks); + expand_phase_edges_admitted += thread_walk_edges_admitted; + *edges_admitted += thread_walk_edges_admitted; + if (thread_walk_frontier_cap_hit) { + // Frontier-cap mid-thread-walk is the same search-abandonment grounds + // as anywhere else - do not spend more of this pass's budget. + *truncated = true; + *frontier_cap_hit = true; + return; + } + if (thread_walk_truncated) { + *truncated = true; + } + } + + // Static-field roots (SomeClass.staticField -> obj) are not reachable via + // IterateOverReachableObjects' root/stack-ref callbacks above - see + // admitStaticFieldRoots()'s own comment - so this pass would otherwise + // never discover an object retained only that way. Best-effort: failures + // here do not truncate the pass, they just mean this sweep found nothing + // new this time around. + // + // Only run the sweep when the loaded-class set has actually changed since + // the last time it completed (same guard shape resolveLoadedClasses() uses + // for its own GetLoadedClasses()-driven scan, and reusing the count that + // call already refreshed via resolveLoadedClasses() earlier this same + // runPass() - see _last_static_field_class_count's own comment). Without + // this, admitStaticFieldRoots() would re-run its own GetLoadedClasses() + // call and a FollowReferences over every loaded class - a stop-the-world + // HeapWalkOperation - on every pass, forever, at the per-second pass + // cadence, even once every loaded class's static fields have already been + // swept and no new class has appeared to introduce new ones. + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk static_sweep_gate " + "resolved=%d swept=%d cursor=%d", + _last_resolved_class_count, _last_static_field_class_count, + _static_field_sweep_cursor); + if (_last_resolved_class_count != _last_static_field_class_count) { + int static_field_edges_admitted = 0; + bool static_field_truncated = false; + bool static_field_frontier_cap_hit = false; + bool static_field_cycle_complete = false; + int static_field_budget = std::max(budget - expand_phase_edges_admitted, 0); + admitStaticFieldRoots(jvmti, jni, _hop_cap, static_field_budget, + &static_field_edges_admitted, &static_field_truncated, + &static_field_frontier_cap_hit, + &static_field_cycle_complete, safepoint_ticks); + expand_phase_edges_admitted += static_field_edges_admitted; + *edges_admitted += static_field_edges_admitted; + if (static_field_truncated) { + *truncated = true; + *frontier_cap_hit = static_field_frontier_cap_hit; + if (static_field_frontier_cap_hit) { + // Frontier-size cap hit while admitting static-field roots is the + // same "grounds to ABANDON the whole search" outcome + // BUDGET_EXHAUSTED/FRONTIER_CAP_HIT handling above gives root + // enumeration - do not spend any more of this pass's budget on the + // ordinary expansion below. + return; + } + } + if (static_field_cycle_complete) { + // The chunk cursor completed a full lap over the loaded-class list + // with no chunk truncating along the way (possibly discovering + // nothing, if every static field seen was already ALREADY_ADMITTED) - + // remember the class count it covered so a later pass with no new + // classes can skip re-running the sweep entirely. Left unset if any + // chunk in the lap truncated (admitStaticFieldRoots() already started + // the next lap immediately in that case) so passes keep retrying + // instead of wrongly treating a still-incomplete sweep as done. + _last_static_field_class_count = _last_resolved_class_count; + } + } + + int expand_edges_admitted = 0; + bool expand_truncated = false; + bool expand_frontier_cap_hit = false; + int remaining_budget = std::max(budget - expand_phase_edges_admitted, 0); + // Give expand its own fresh deadline so the static-field sweep's + // FollowReferences calls don't eat expand's time. Each sub-operation + // (sweep, expand, rotation) gets its own _effective_pause_target_ms + // wall-clock budget — the cumulative rate is still capped by the pass + // cadence (effectiveCadenceNs). See q-safepoint-budget-model. + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + expandFrontier(jvmti, jni, _hop_cap, remaining_budget, + &expand_edges_admitted, &expand_truncated, + &expand_frontier_cap_hit, safepoint_ticks); + expand_phase_edges_admitted += expand_edges_admitted; + *edges_admitted += expand_edges_admitted; + *truncated = *truncated || expand_truncated; + *frontier_cap_hit = expand_frontier_cap_hit; + TEST_LOG_SUMMARY("ReferenceChainTracker::runPassManualWalk expand_phase " + "edges_admitted=%d truncated=%d frontier_cap_hit=%d " + "remaining_budget=%d", + expand_edges_admitted, (int)expand_truncated, + (int)expand_frontier_cap_hit, remaining_budget); + + // Note: unlike a hard truncation during root/stack-ref enumeration or the + // static-field sweep above (which return early - the pass never even + // reached ordinary expansion), a truncated ordinary expansion does NOT + // skip rotation below: rotation runs on its own reserved slice of budget + // (see rotation_reserved_budget's own comment above) precisely because + // ordinary expansion truncates on nearly every pass under a sustained + // fast-growing backlog, and that is exactly the situation - a mutable + // field reassigned out from under an already-EXPANDED entry - rotation + // exists to correct. + + // Three-tier bounded rotating re-expansion (design doc's closing section, + // extended - see each collector's own comment for why it + // exists as its own tier): re-walk a bounded, rotating subset of + // already-EXPANDED entries so mutations to an already-expanded object's + // fields - a durable root discovered elsewhere for a stale attribution, or + // a mutable collection field reassigned out from under a container object + // - get a chance to be observed on a later pass. Runs after the ordinary + // expansion above so it only ever spends whatever budget that left + // unused, plus its own reserved slice. Ordered highest-value/cheapest + // first: root-attribution re-verification (small, bounded population), + // then the leak-accumulation growth-catching tier (also small, bounded, + // and the one that actually targets this leak shape), then the + // unprioritized whole-table fallback last. + std::vector rotation_tags = + collectStaleRootKindEntriesForRotation(ROOT_KIND_ROTATION_BUDGET); + std::vector leak_accumulation_tags = + collectLeakAccumulationCandidatesForRotation( + LEAK_ACCUMULATION_ROTATION_BUDGET); + // Also re-walk a bounded, rotating subset of EXPANDED entries regardless + // of root attribution: a mutable field reassigned since an + // object's one-time expansion - e.g. HashMap.table on resize - is + // otherwise never observed again, silently orphaning everything only + // reachable through the field's current value. See + // collectStaleExpandedEntriesForRotation()'s own comment. + std::vector stale_expanded_tags = + collectStaleExpandedEntriesForRotation(STALE_EXPANDED_ROTATION_BUDGET); + // Candidate-scoped reach, prong 2: root-attached static holders are + // descend-walked directly (see collectStaticFieldAnchorsForRotation()/ + // walkStaticFieldAnchors()'s own comments) - not pushed onto the priority + // lane, so they are independent of the queue tiers above. The at-risk + // FIFO (B', _static_anchor_fifo's declaration comment) is drained BEHIND + // the collector's selection: the collector's small root-attached cohort + // walks first (a single-referrer static holder like the LEAK_BUFFER + // wrapper lives there - it never demotes, so it can never be at-risk), + // and a truncated pass falls on the FIFO's at-risk suffix, which the + // requeue path below protects - observed on the pod (round 10) that the + // reverse order starved the collector's picks in ~60% of passes + // (walked=6-16 of selected=20) against a cap-pinned at-risk flood. + // Classify any not-yet-shaped anchor classes (up to + // ANCHOR_SHAPE_RECONCILE_BUDGET per pass, one GOTW call) BEFORE the + // collector runs, so this pass's tiering sees as much of the container + // cohort as possible. Runs on the engine thread with JNI available, + // outside heap callbacks. + reconcileAnchorClassShapes(jvmti, jni); + std::vector static_anchor_tags = + collectStaticFieldAnchorsForRotation(STATIC_ANCHOR_ROTATION_BUDGET); + std::vector static_anchor_fifo_drained; + drainStaticAnchorFifo(STATIC_ANCHOR_FIFO_DRAIN, static_anchor_fifo_drained); + for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { + static_anchor_tags.push_back(at_risk.tag); + } + if (rotation_tags.empty() && leak_accumulation_tags.empty() && + stale_expanded_tags.empty() && static_anchor_tags.empty()) { + return; + } + // rotation_reserved_budget + max(budget - expand_phase_edges_admitted, 0) is + // exactly expand_budget - expand_phase_edges_admitted: budget already IS + // expand_budget - rotation_reserved_budget (above), and expand_phase_edges_ + // admitted can never exceed budget (the static-field sweep and ordinary + // expandFrontier() calls above are both capped to budget-derived slices), + // so the max() is never actually needed to avoid going negative. Folding + // rotation_reserved_budget back into expand_budget here - rather than + // subtracting it out and then adding it back - says directly what this + // value is: whatever of the whole pass's budget the phases above didn't + // spend. + int rotation_budget = expand_budget - expand_phase_edges_admitted; + int rotation_edges_admitted = 0; + bool rotation_truncated = false; + // Give rotation its own fresh deadline, same as expand above. + _pass_deadline_ns = _effective_pause_target_ms > 0 + ? OS::nanotime() + (u64)_effective_pause_target_ms * 1000000ULL + : 0; + bool rotation_frontier_cap_hit = false; + // Prong 2 static-anchor descend walks run FIRST inside rotation's slice: + // they are the highest-value rotation work (bounded, targeted, and the + // only rotation tier that can reach a collection-shaped static holder's + // internals in one pass), and their edges draw down the same rotation + // budget the queue-tier batch below uses - a pass whose anchor walks admit + // the holder's whole internal structure needs less one-hop rotation work, + // not more. + if (!static_anchor_tags.empty() && rotation_budget > 0) { + int static_anchor_edges_admitted = 0; + bool static_anchor_truncated = false; + bool static_anchor_frontier_cap_hit = false; + std::vector static_anchor_unwalked; + walkStaticFieldAnchors(jvmti, jni, static_anchor_tags, rotation_budget, + &static_anchor_edges_admitted, + &static_anchor_truncated, + &static_anchor_frontier_cap_hit, safepoint_ticks, + &static_anchor_unwalked); + rotation_edges_admitted += static_anchor_edges_admitted; + rotation_budget -= static_anchor_edges_admitted; + *truncated = *truncated || static_anchor_truncated; + // B' requeue: resolved-but-unwalked anchors that came from this pass's + // FIFO drain go back to the FIFO front, order-preserving, so a pass + // whose budget died mid-batch walks them first next pass instead of + // waiting for the next sweep lap's re-push. Collector-sourced un-walked + // anchors are deliberately dropped from this - their retention is the + // collector cursor's own. Frontier lookups filter entries that died + // between selection and the walk (requeueing a dead tag would only + // re-drop it). The scan is drained x unwalked (<= 16 x <= 20), well + // under the small-set linear-scan cutoff. Entries keep their klass so + // the requeue re-increments the per-class occupancy exactly. + if (!static_anchor_unwalked.empty() && + !static_anchor_fifo_drained.empty()) { + std::vector static_anchor_requeue; + for (jlong tag : static_anchor_unwalked) { + FrontierEntry entry{}; + if (!_frontier->lookup(tag, &entry)) { + continue; + } + for (const AtRiskAnchor &at_risk : static_anchor_fifo_drained) { + if (tag == at_risk.tag) { + static_anchor_requeue.push_back(at_risk); + break; + } + } + } + if (!static_anchor_requeue.empty()) { + requeueStaticAnchorFifoFront(static_anchor_requeue); + } + } + if (static_anchor_frontier_cap_hit) { + *frontier_cap_hit = true; + return; + } + } + // expandFrontier() SETS (does not add into) its edges output - see its + // entry - so the anchor walks' edges are kept in a separate counter and + // summed here. + int queue_tier_edges_admitted = 0; + expandFrontier(jvmti, jni, _hop_cap, rotation_budget, + &queue_tier_edges_admitted, &rotation_truncated, + &rotation_frontier_cap_hit, safepoint_ticks); + rotation_edges_admitted += queue_tier_edges_admitted; + *edges_admitted += rotation_edges_admitted; + // OR, not overwrite: the ordinary expand phase above may have already set + // these to true (real truncation/cap-hit left in _pending_expand), and a + // rotation batch that happens to finish cleanly must not erase that - + // has_pending_frontier (runPass()) and the FRONTIER_CAP abandon check both + // read these as "did any of this pass's sub-phases truncate/cap-hit", not + // just the last one that ran. + *truncated = *truncated || rotation_truncated; + *frontier_cap_hit = *frontier_cap_hit || rotation_frontier_cap_hit; + + // End-of-pass drain for the heap callback's deferred resolved-chain + // invalidations (improveChain/re-parent/root-upgrade evictions recorded + // during the walks): applying them here keeps _resolved_chains_lock + // traffic outside the FollowReferences pauses entirely, and ensures the + // next pollWatchedTargets() never rebuilds from a stale cached chain. + drainPendingChainInvalidations(); +} + +// --------------------------------------------------------------------------- +// Incremental resumption across passes. +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::markAllFrontierExpanded() { + while (!_priority_expand.empty()) { + _frontier->markExpanded(_priority_expand.front()); + _priority_expand.pop_front(); + } + _priority_expand_set.clear(); + while (!_pending_expand.empty()) { + _frontier->markExpanded(_pending_expand.front()); + _pending_expand.pop_front(); + } +} + +void ReferenceChainTracker::expandFrontier(jvmtiEnv *jvmti, JNIEnv *jni, + int hop_cap, int budget, + int *edges_admitted, + bool *truncated, + bool *frontier_cap_hit, + u64 *safepoint_ticks) { + assert(!t_inGCCallback && + "GetObjectsWithTags/FollowReferences are JVMTI Heap-category calls " + "and must not be made from GarbageCollectionStart/Finish"); + + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx.hop_cap = hop_cap; + ctx.budget = budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + + // ARRAY-HOLDER BATCHING: expand a whole batch of boundary objects with ONE + // FollowReferences(initial_object=holder_array) call per BFS level, instead + // of one FollowReferences PER frontier entry. batch_tags gates + // heapReferenceCallback() to a single hop (see its own comment). This is + // the unconditional default expansion path (runPass()'s only non-fallback + // walk), not a prototype relative to anything else still in the codebase. + std::unordered_set batch_tags; + ctx.batch_tags = &batch_tags; + + jvmtiHeapCallbacks callbacks; + memset(&callbacks, 0, sizeof(callbacks)); + callbacks.heap_reference_callback = heapReferenceCallback; + + // java/lang/Object element type for the transient frontier-holder array. + // Cached across calls on this same (attached) JNIEnv rather than re-resolved + // via a fresh FindClass() every call - expandFrontier() runs roughly once + // per BFS-thread wake for the tracker's lifetime, and the class never + // changes, so a per-pass class-loader lookup is unnecessary churn. Without + // a JNIEnv (some test seams) the array-holder path cannot run; a JNIEnv + // change (fresh attach) invalidates the cache since the previous call's + // local ref is only guaranteed valid for that attach's lifetime. + if (jni != nullptr && _cached_object_class == nullptr) { + jclass local = jni->FindClass("java/lang/Object"); + if (!jniExceptionCheck(jni) && local != nullptr) { + _cached_object_class = (jclass)jni->NewGlobalRef(local); + } + if (local != nullptr) { + jni->DeleteLocalRef(local); + } + } + jclass object_class = _cached_object_class; + + bool progress = true; + // FAIR-SHARE DRAIN: alternate batches between _priority_expand and + // _pending_expand whenever both are non-empty (priority still takes the + // first batch of each call). The original strict priority-first drain + // starved the ordinary backlog whenever rotation's inflow + // (~STALE_EXPANDED_ROTATION_BUDGET+ROOT_KIND_ROTATION_BUDGET per pass) + // exceeded the deadline-bounded drain (~2-3 GetObjectsWithTags calls per + // phase) - observed live on hotdog: _priority_expand grew 39k->103k in 20 + // minutes while the BFS's own _pending_expand (66k entries) was never + // drained by a single batch, freezing all new-territory crawl. + // Alternation guarantees the ordinary frontier at least every other + // batch regardless of queue depths; an empty lane falls back to the + // other one. The toggle is the _expand_lane_prefer_priority MEMBER + // (not a local of this invocation): the phase deadlines bound a typical + // invocation to a single batch, so a per-invocation reset made priority + // win every invocation - observed live on hotdog with round 3's build, + // where _pending_expand GREW 109k->113k across 260 passes while every + // gotw call drained the priority lane's stale re-walks (edges=0). + while (!ctx.truncated && progress && object_class != nullptr) { + // Wall-clock deadline check per iteration: GetObjectsWithTags runs OUTSIDE + // any FollowReferences callback, so heapReferenceCallback()'s amortized + // deadline check never sees its cost. Measured live on hotdog: a + // collapsed batch (batch=2) let ~1400 unchecked GetObjectsWithTags calls + // (~20ms each) run in one expand phase, spending 10.4s of CPU and ~30s of + // wall time in a single pass. Checking here bounds each phase to + // _effective_pause_target_ms regardless of batch health. + if (_pass_deadline_ns != 0 && OS::nanotime() >= _pass_deadline_ns) { + ctx.truncated = true; + break; + } + progress = false; + + // Alternate lanes (see FAIR-SHARE DRAIN above); priority still goes + // first so a rotation-selected parent's re-discovery keeps its + // head-of-queue property, but no lane can monopolize the drain. + bool from_priority; + if (_priority_expand.empty()) { + from_priority = false; + } else if (_pending_expand.empty()) { + from_priority = true; + } else { + from_priority = _expand_lane_prefer_priority; + _expand_lane_prefer_priority = !_expand_lane_prefer_priority; + } + std::deque &source = + from_priority ? _priority_expand : _pending_expand; + ctx.admit_priority = from_priority; + if (source.empty()) { + break; // nothing pending in either lane + } + + // SELF-CALIBRATING ADAPTIVE BATCH SIZE for GetObjectsWithTags. + // GetObjectsWithTags iterates the whole JVMTI tag map per call, so its + // cost has a batch-independent floor that grows with the frontier + // (measured live: ~20ms at a 225k-entry map regardless of batch_size). + // Calibrating batch_size from a per-tag EMA collapses in that regime + // (small batch inflates per-tag cost, which shrinks the batch further — + // observed live driving batch from ~400 to 2). Instead, AIMD directly on + // batch size against the measured per-CALL time vs GOTW_CPU_BUDGET_NS — + // see _gotw_batch_size's own comment. + // + // Still capped at `budget` and `_budget` for the original reasons + // (first-pass budget can be far larger than the backlog; a single + // huge batch risks JNI local-capacity/OOM with zero progress). + size_t gotw_batch_size = + _gotw_batch_size != 0 ? _gotw_batch_size : GOTW_INITIAL_BATCH_SIZE; + size_t batch_size = std::min( + source.size(), + std::min((size_t)std::max(std::min(budget, _budget), 1), + gotw_batch_size)); + std::vector candidate_tags(source.begin(), + source.begin() + batch_size); + + // Resolve this batch's live boundary objects. GetObjectsWithTags iterates + // the whole tag map, but does so under a no-safepoint mutex on this + // (Java) thread - it is NOT a stop-the-world VM operation, unlike the + // FollowReferences below (jvmtiTagMap.cpp: get_objects_with_tags takes + // Mutex::_no_safepoint_check_flag and calls entry_iterate directly, + // whereas follow_references does VMThread::execute()). + jint resolved_count = 0; + jobject *resolved_objects = nullptr; + jlong *resolved_tags = nullptr; + u64 gotw_start_ns = OS::nanotime(); + jvmtiError resolve_err = jvmti->GetObjectsWithTags( + (jint)candidate_tags.size(), candidate_tags.data(), &resolved_count, + &resolved_objects, &resolved_tags); + u64 gotw_elapsed_ns = OS::nanotime() - gotw_start_ns; + // Self-calibrate (PROPORTIONAL batch control): update the EMA of + // PER-CALL elapsed time, then scale the batch so ONE call fills the + // remaining wall-clock window. This replaces the earlier per-call AIMD + // (fixed budget, halve/add-64): measured live on hotdog, the tag map + // grew until the per-call floor alone (~27ms at a 243k-entry map) + // exceeded the fixed 25ms budget, so AIMD ratcheted to GOTW_MIN_BATCH + // and stayed there (batch=8 forever) even though batch=72 cost only + // +36% for 9x the objects - the floor-dominated regime in which a + // BIGGER batch is the right move, and only a proportion against the + // remaining deadline can see that. See _gotw_batch_size's own + // comment for the full history (incl. the earlier per-tag collapse). + if (batch_size > 0 && gotw_elapsed_ns > 0) { + if (_gotw_ema_call_ns == 0) { + _gotw_ema_call_ns = gotw_elapsed_ns; + } else { + _gotw_ema_call_ns = _gotw_ema_call_ns * 4 / 5 + gotw_elapsed_ns / 5; + } + u64 now_ns = OS::nanotime(); + u64 window_ns = + gotwWindowNs( + _pass_deadline_ns != 0 && _pass_deadline_ns > now_ns + ? _pass_deadline_ns - now_ns + : 0, + source.size()); + // window_ns / ema_call_ns == how many such calls fit the window; + // scaling the CURRENT calibration batch by that ratio sizes the next + // call to consume the whole window in one go. Extrapolate from the + // stored _gotw_batch_size (the intended size), not from batch_size: + // batch_size is capped by the lane depth (min(source.size(), ...)), + // and a shallow lane would calibrate the stored size toward its own + // depth even though the stored size is what the next deep-lane call + // will use. Integer division biases the next batch slightly small - + // safe (an under-filled window just runs a second call; an + // over-filled one overruns the deadline). + size_t calib_batch = + _gotw_batch_size != 0 ? _gotw_batch_size : GOTW_INITIAL_BATCH_SIZE; + size_t next_batch = (size_t)((u64)calib_batch * window_ns / + std::max(_gotw_ema_call_ns, 1ULL)); + _gotw_batch_size = std::min(std::max(next_batch, GOTW_MIN_BATCH), + GOTW_MAX_BATCH); + } + if (resolve_err != JVMTI_ERROR_NONE) { + // GetObjectsWithTags failed: per the JVMTI contract the output arrays + // are NULL and the count is 0 on error, but do not rely on the count - + // a stale/garbage resolved_count here would drive the shared cleanup + // block at the loop bottom into DeleteLocalRef on garbage pointers. + // Clamping keeps the break path (which skips the per-iteration + // cleanup) safe: the null-guarded Deallocate calls there are no-ops. + resolved_count = 0; + ctx.truncated = true; + break; + } + + std::unordered_map live; + for (jint i = 0; i < resolved_count; i++) { + live[resolved_tags[i]] = resolved_objects[i]; + } + + // Build the frontier-holder array from the live boundary objects and + // record their tags so heapReferenceCallback() descends into exactly + // these (one hop). + batch_tags.clear(); + jobjectArray holder = nullptr; + if (resolved_count > 0) { + jint capacity_err = jni->EnsureLocalCapacity(resolved_count + 16); + // Check the exception FIRST (jniExceptionCheck clears it): a + // short-circuited `capacity_err < 0 || jniExceptionCheck(jni)` would + // skip the check when capacity itself failed, leaving a pending + // OutOfMemoryError to survive into the next JNI call on this + // long-lived BFS-thread JNIEnv (JNI spec: undefined behavior with a + // pending exception across ordinary JNI calls). + bool capacity_exc = jniExceptionCheck(jni); + if (capacity_err < 0 || capacity_exc) { + // Could not guarantee local-ref headroom for this batch - treat like + // any other batch-level failure below (JVMTI error / OOM building the + // holder array): retry this batch on a later pass rather than + // proceeding into NewObjectArray with no capacity guarantee. + ctx.truncated = true; + } else { + holder = jni->NewObjectArray(resolved_count, object_class, nullptr); + if (jniExceptionCheck(jni)) { + // OutOfMemoryError building the holder array (or any other + // exception NewObjectArray raised) left `holder` null; make sure + // the pending exception does not survive into the next JNI call + // below or the next expandFrontier() invocation on this same + // long-lived BFS-thread JNIEnv (JNI spec: undefined behavior with + // a pending exception across ordinary JNI calls). + holder = nullptr; + } + if (holder != nullptr) { + for (jint i = 0; i < resolved_count; i++) { + jni->SetObjectArrayElement(holder, i, resolved_objects[i]); + if (jniExceptionCheck(jni)) { + // e.g. an array-store-class failure. Abort building this + // batch's holder rather than handing a partially-populated + // array (with a just-cleared pending exception) to + // FollowReferences. + ctx.truncated = true; + break; + } + batch_tags.insert(resolved_tags[i]); + } + } + if (holder == nullptr) { + // NewObjectArray failed (OOM/local-ref exhaustion) - the + // FollowReferences call below (which would have discovered this + // batch's children) never runs. Falling through to the + // mark-EXPANDED-and-dequeue path further down would silently and + // permanently drop these still-undiscovered children, so this must + // be treated exactly like a failed FollowReferences/JVMTI call: + // retry the batch on a later pass instead. + ctx.truncated = true; + } else if (!ctx.truncated) { + // A single FollowReferences over the holder array expands this whole + // BFS level in one stop-the-world HeapWalkOperation (instead of one + // per frontier entry). initial_object=holder means the traversal + // starts from the array only (never enumerates roots / the whole + // heap); heapReferenceCallback() returns "descend" for the array's + // elements (the boundary objects, in batch_tags) and "no descend" for + // their children, so exactly one hop past the boundary is explored. + ctx._last_visited_batch_tag = 0; // reset rolling cursor + u64 follow_start_ticks = TSC::ticks(); + jvmtiError follow_err = + jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); + *safepoint_ticks += TSC::ticks() - follow_start_ticks; + if (follow_err != JVMTI_ERROR_NONE) { + ctx.truncated = true; + } + } + } + } + + if (!ctx.truncated) { + // The whole batch had all its direct children admitted this level: + // dead entries are pruned, live ones are marked EXPANDED, and all are + // popped off the front. New children were appended to the back by + // admitObject() and become the next level's batch. + for (jlong tag : candidate_tags) { + if (live.find(tag) == live.end()) { + _frontier->clear(tag); + } else { + _frontier->markExpanded(tag); + } + source.pop_front(); + } + progress = true; + } else if (ctx._last_visited_batch_tag != 0) { + // ROLLING RESUME: FollowReferences truncated mid-batch, but we know + // which batch entry was being visited when it stopped (tracked by + // the callback's batch_tags descent-gate). Pop entries that were + // fully processed BEFORE that entry (mark live ones EXPANDED, clear + // dead ones), and leave the partially-processed entry and everything + // after it at the front of the source queue for the next pass to + // retry. Same resumable-cursor pattern as admitStaticFieldRoots()'s + // sweep cursor — avoids re-walking already-expanded entries (and + // re-paying GetObjectsWithTags's O(tag_map × batch) cost for them) + // on every retry. + // + // The partially-visited entry (at _last_visited_batch_tag) stays: + // some of its children may have been admitted before the truncation, + // and the rest are discovered on retry (admitObject is idempotent — + // already-admitted children return ALREADY_ADMITTED). + for (size_t i = 0; i < candidate_tags.size(); i++) { + if (candidate_tags[i] == ctx._last_visited_batch_tag) { + break; // stop at the partially-visited entry + } + jlong tag = candidate_tags[i]; + if (live.find(tag) == live.end()) { + _frontier->clear(tag); + } else { + _frontier->markExpanded(tag); + } + source.pop_front(); + } + } + // else truncated with no batch entry visited (e.g. GetObjectsWithTags + // error, holder allocation failure, or truncation before the first + // batch entry was reached): leave the entire batch at the front of the + // source queue for a later pass to retry, same as before. + + if (from_priority) { + // This batch popped entries off _priority_expand's front (or, on + // truncation, was left untouched) - re-derive the membership index + // from the deque's current contents either way so + // isQueuedForRotation() stays exact for the rotation collectors that + // run later in this same pass. A full rebuild is <= + // PRIORITY_EXPAND_CAP inserts, a few microseconds against the ~20ms + // GetObjectsWithTags call this batch already paid + // (PriorityExpandSet's own comment, referenceChains.h). + _priority_expand_set.rebuildFrom(_priority_expand); + } + + if (holder != nullptr) { + jni->DeleteLocalRef(holder); + } + if (jni != nullptr) { + for (jint i = 0; i < resolved_count; i++) { + jni->DeleteLocalRef(resolved_objects[i]); + } + } + if (resolved_objects != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_objects); + } + if (resolved_tags != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_tags); + } + } + + // object_class is NOT deleted here - it is now cached in + // _cached_object_class and reused across calls on this same JNIEnv (see + // above), not a per-call local ref. + + if (!ctx.truncated && jni != nullptr && object_class == nullptr && + (!_pending_expand.empty() || !_priority_expand.empty())) { + // FindClass("java/lang/Object") failed for this (attached) JNIEnv, so + // the batching loop above never ran even though pending frontier work + // remains. Report truncated rather than leaving *truncated false: the + // caller (runPassManualWalk()/runPass()) treats false as "no pending + // frontier work", which would falsely mark the search + // SearchState::COMPLETED instead of retrying - directly contradicting + // this subsystem's documented "no silent truncation" requirement (see + // SearchAbandonReason's header comment). + ctx.truncated = true; + } + + *edges_admitted = ctx.edges_admitted; + *truncated = ctx.truncated; + *frontier_cap_hit = ctx.frontier_cap_hit; +} + +void ReferenceChainTracker::admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, + int hop_cap, int budget, + int *edges_admitted, + bool *truncated, + bool *frontier_cap_hit, + bool *cycle_complete, + u64 *safepoint_ticks) { + assert(!t_inGCCallback && + "GetLoadedClasses/FollowReferences are JVMTI Heap-category calls " + "and must not be made from GarbageCollectionStart/Finish"); + *edges_admitted = 0; + *truncated = false; + *frontier_cap_hit = false; + *cycle_complete = false; + + if (jni == nullptr) { + // No JNIEnv to build the holder array on (some test seams) - see + // expandFrontier()'s own identical guard. Best-effort sweep: nothing + // discovered this call, not this pass's own truncation. + return; + } + + jint class_count = 0; + jclass *classes = nullptr; + jvmtiError classes_err = jvmti->GetLoadedClasses(&class_count, &classes); + if (classes_err != JVMTI_ERROR_NONE) { + return; + } + if (class_count <= 0) { + if (classes != nullptr) { + jvmti->Deallocate((unsigned char *)classes); + } + return; + } + + // GetLoadedClasses() gives no ordering guarantee across separate calls, so + // the cursor below is only meaningful as an index into THIS call's array - + // reprioritize it every call rather than trying to cache an ordering. + // Application/library classes (any non-bootstrap classloader) are moved to + // the front so a chunked sweep (below) reaches a likely leak source within + // its first several chunks instead of only after every JDK/platform class + // (typically the majority of a real JVM's loaded-class count) has been + // swept first. In-place two-way partition, no extra allocation. + jint app_boundary = 0; + for (jint i = 0; i < class_count; i++) { + jobject loader = nullptr; + jvmtiError loader_err = jvmti->GetClassLoader(classes[i], &loader); + bool is_app_class = (loader_err == JVMTI_ERROR_NONE) && (loader != nullptr); + if (loader != nullptr) { + jni->DeleteLocalRef(loader); + } + if (is_app_class) { + if (i != app_boundary) { + std::swap(classes[i], classes[app_boundary]); + } + app_boundary++; + } + } + + if (_static_field_sweep_cursor >= class_count) { + // Loaded-class count shrank since the last chunk (classes unloaded) - + // restart the lap rather than reading out of range. + _static_field_sweep_cursor = 0; + _static_field_sweep_cycle_truncated = false; + } + jint chunk_start = _static_field_sweep_cursor; + jint chunk_end = + std::min(chunk_start + STATIC_FIELD_SWEEP_CHUNK_CLASSES, class_count); + jint chunk_count = chunk_end - chunk_start; + + // Same java/lang/Object element-type cache expandFrontier() uses for its + // own frontier-holder array - shared across both call sites on this same + // attached JNIEnv rather than a second FindClass() per pass. + if (_cached_object_class == nullptr) { + jclass local = jni->FindClass("java/lang/Object"); + if (!jniExceptionCheck(jni) && local != nullptr) { + _cached_object_class = (jclass)jni->NewGlobalRef(local); + } + if (local != nullptr) { + jni->DeleteLocalRef(local); + } + } + jclass object_class = _cached_object_class; + + if (object_class == nullptr || + jni->EnsureLocalCapacity(class_count + 16) < 0 || + jniExceptionCheck(jni)) { + for (jint i = 0; i < class_count; i++) { + jni->DeleteLocalRef(classes[i]); + } + jvmti->Deallocate((unsigned char *)classes); + return; + } + + jobjectArray holder = jni->NewObjectArray(chunk_count, object_class, nullptr); + if (jniExceptionCheck(jni)) { + // OutOfMemoryError (or any other exception) building the holder - + // clear it rather than let it survive into the DeleteLocalRef() calls + // below (JNI spec: undefined behavior with a pending exception across + // ordinary JNI calls), same as expandFrontier()'s identical case. + holder = nullptr; + } + if (holder != nullptr) { + // Fill in REVERSE chunk order: holder[0] = classes[chunk_end-1], ..., + // holder[chunk_count-1] = classes[chunk_start]. HotSpot's + // FollowReferences visits the initial_object (the holder array) by + // pushing it on a LIFO visit_stack and popping (jvmtiTagMap.cpp: + // iterate_over_array pushes elements 0..n-1 in order, the while-loop + // pops LIFO), so classes are descended in REVERSE holder order. + // Reversing the fill makes the descent visit classes in ASCENDING + // original index order (chunk_start first) on HotSpot, which spreads + // the chunk's admit budget in the app-classes-first priority order. + // The truncation cursor below no longer depends on this ordering (it + // redoes the whole chunk), so a JVM whose visit order differs only + // gets a different in-chunk priority, never a wrong resume. + for (jint i = 0; i < chunk_count; i++) { + jni->SetObjectArrayElement(holder, i, classes[chunk_end - 1 - i]); + if (jniExceptionCheck(jni)) { + holder = nullptr; + break; + } + } + } + + // GetLoadedClasses() returned a local ref for every class regardless of + // chunk selection - free all of them here, not just the chunk. + for (jint i = 0; i < class_count; i++) { + jni->DeleteLocalRef(classes[i]); + } + jvmti->Deallocate((unsigned char *)classes); + + if (holder == nullptr) { + // OOM/local-ref exhaustion/array-store failure - skip this pass's sweep + // rather than treating it like the manual walk's own truncation (see + // this method's own header comment). + return; + } + + PassContext ctx; + ctx.tracker = this; + ctx.frontier = _frontier; + ctx.hop_cap = hop_cap; + ctx.budget = budget; + ctx.edges_admitted = 0; + ctx.truncated = false; + ctx.frontier_cap_hit = false; + // Empty (not null) batch_tags forces heapReferenceCallback() to stop at + // exactly one hop past each class - see this method's own header comment + // for why a deeper descent here would reintroduce the whole-graph + // FollowReferences cost the array-holder batching design otherwise avoids. + std::unordered_set empty_batch_tags; + ctx.batch_tags = &empty_batch_tags; + // Lets heapReferenceCallback() walk past the holder->class seed edge (see + // PassContext::static_field_seed's own comment) so this sweep actually + // reaches each class's static fields instead of stopping at the + // negative-tagged class object itself. + ctx.static_field_seed = true; + // Per-class non-STATIC_FIELD admission cap (see PassContext::_class_other_cap's + // own comment). STATIC_FIELD edges are always admitted; non-static edges + // (CONSTANT_POOL, INTERFACE, SUPERCLASS, CLASS_LOADER, ...) are admitted + // up to this many per class per lap, then dropped for the rest of that + // class. 32 covers a typical class's full constant-pool/interface set; + // outlier classes are bounded so they cannot blow the chunk's deadline. + ctx._class_other_cap = STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS; + + jvmtiHeapCallbacks callbacks; + memset(&callbacks, 0, sizeof(callbacks)); + callbacks.heap_reference_callback = heapReferenceCallback; + u64 follow_start_ticks = TSC::ticks(); + jvmtiError follow_err = + jvmti->FollowReferences(0, nullptr, holder, &callbacks, &ctx); + *safepoint_ticks += TSC::ticks() - follow_start_ticks; + jni->DeleteLocalRef(holder); + if (follow_err != JVMTI_ERROR_NONE) { + return; + } + + *edges_admitted = ctx.edges_admitted; + *truncated = ctx.truncated; + *frontier_cap_hit = ctx.frontier_cap_hit; + + if (ctx.truncated) { + _static_field_sweep_cycle_truncated = true; + // Resumable cursor: instead of skipping to chunk_end (losing every + // class after the interruption point for the rest of this lap), redo + // the chunk on the next pass. An earlier refinement resumed at + // chunk_start + classes-visited - 1, which assumes the abort position + // maps back to ascending original index order - a property of + // HotSpot's current LIFO FollowReferences visit order that no JVMTI + // implementation guarantees (and this is shared code, per the + // project's JVM-support rules). Redoing the chunk is order- + // independent: re-walked classes hit ALREADY_ADMITTED cheaply (the + // per-class quota bounds their non-static edges), and the chunk is + // bounded by STATIC_FIELD_SWEEP_CHUNK_CLASSES. + _static_field_sweep_cursor = chunk_start; + } else { + // Full advance: every class in the chunk was processed. + _static_field_sweep_cursor = chunk_end; + } + if (_static_field_sweep_cursor >= class_count) { + *cycle_complete = !_static_field_sweep_cycle_truncated; + _static_field_sweep_cursor = 0; + _static_field_sweep_cycle_truncated = false; + } +} + +bool ReferenceChainTracker::releaseSearchTags(jvmtiEnv *jvmti, JNIEnv *jni) { + assert(!t_inGCCallback && + "GetObjectsWithTags is a JVMTI Heap-category call and must not be " + "made from GarbageCollectionStart/Finish"); + if (jvmti == nullptr || _frontier == nullptr) { + return true; // nothing to release + } + + jlong scan_limit = _frontier->size(); + std::vector live_tags; + for (jlong tag = 1; tag <= scan_limit; tag++) { + FrontierEntry entry{}; + if (_frontier->lookup(tag, &entry) && + entry.state != FrontierEntryState::ABANDONED) { + live_tags.push_back(tag); + } + } + if (live_tags.empty()) { + return true; + } + + jint resolved_count = 0; + jobject *resolved_objects = nullptr; + jlong *resolved_tags = nullptr; + if (jvmti->GetObjectsWithTags((jint)live_tags.size(), live_tags.data(), + &resolved_count, &resolved_objects, + &resolved_tags) != JVMTI_ERROR_NONE) { + // GetObjectsWithTags() itself failed (e.g. JVMTI_ERROR_OUT_OF_MEMORY): + // we do NOT know which, if any, of live_tags are still live objects, so + // do not mark any of them ABANDONED here - doing so while their JVMTI + // tag might still be set would let a restarted search's nextTag() + // sequence eventually reissue the same numeric tag to a brand-new + // object, corrupting FrontierTable's tag-uniqueness invariant (see this + // method's own header comment). Report failure so the caller retries + // this same batch later instead of proceeding to restart. + Counters::increment(REFERENCE_CHAIN_TAG_RELEASE_FAILED); + Log::warn("ReferenceChains: GetObjectsWithTags failed while releasing " + "%zu search tag(s); will retry before allowing a search " + "restart", + live_tags.size()); + return false; + } + + for (jint i = 0; i < resolved_count; i++) { + // clearTag() rather than a raw SetTag() call - reuses the + // same helper (and its GC-callback self-consistency assert) tagObject/ + // getTag already go through. + clearTag(jvmti, resolved_objects[i]); + if (jni != nullptr) { + jni->DeleteLocalRef(resolved_objects[i]); + } + } + if (resolved_objects != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_objects); + } + if (resolved_tags != nullptr) { + jvmti->Deallocate((unsigned char *)resolved_tags); + } + // Tags that failed to resolve above are already dead (JVMTI forgot them + // with their object) - nothing to release, just mark the record ABANDONED + // below like every other entry this search owned. Only reached once + // GetObjectsWithTags() itself succeeded, so every live_tags entry has now + // either been resolved-and-cleared or confirmed dead. + for (jlong tag : live_tags) { + _frontier->clear(tag); + } + return true; +} + +bool ReferenceChainTracker::runPass(jvmtiEnv *jvmti, JNIEnv *jni, + bool *out_truncated) { + if (!_enabled || jvmti == nullptr || _frontier == nullptr) { + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass early-exit: enabled=%d jvmti=%p frontier=%p", + _enabled, (void *)jvmti, (void *)_frontier); + return false; + } + + if (_search_state != SearchState::RUNNING) { + // The search already reached a terminal outcome - nothing left for + // another pass to do until shouldRunPass() decides to restartSearch() + // (this class's header comment), which flips _search_started back to + // false before this method is called again. If a prior terminal-state + // transition's releaseSearchTags() call failed, retry it here rather + // than leaving _tags_released false forever - shouldRunPass() refuses + // to restart the search until this succeeds (see _tags_released's own + // comment), so this is the only remaining call site that can make + // progress on the retry. + if (!_tags_released) { + _tags_released = releaseSearchTags(jvmti, jni); + } + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass no-op: searchState=%d already terminal " + "tagsReleased=%d", + (int)_search_state, _tags_released); + if (out_truncated != nullptr) { + *out_truncated = false; + } + return true; + } + + resolveLoadedClasses(jvmti, jni); + + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass starting JVMTI walk: " + "search_started=%d frontierSize=%zu", + _search_started, _frontier != nullptr ? _frontier->size() : (size_t)0); + + int edges_admitted = 0; + bool truncated = false; + bool frontier_cap_hit = false; + // NOTE: no jvmtiError plumbing - runPassManualWalk() is void and reports + // its failures via the truncated/frontier_cap_hit out-params (the walk + // phases map JVMTI errors onto those flags internally). An earlier draft + // declared an `err` variable that was unconditionally assigned + // JVMTI_ERROR_NONE after the walk and returned as `err == + // JVMTI_ERROR_NONE` - dead logic that made runPass() unconditionally + // return true; the return value now derives from the state that actually + // carries the outcome (see the return below). + // Whole-call wall-clock duration of runPassManualWalk() below - includes + // root/stack-ref enumeration dispatch, frontier-table bookkeeping, and + // rotation-candidate collection, in addition to the actual in-safepoint + // JVMTI calls. Used only to derive non_safepoint_ticks below for + // _cpu_pain_budget; NOT fed to updatePacing()/_pause_pid directly (see + // safepoint_ticks below for that). Measured via TSC::ticks() rather than + // OS::nanotime(), matching this codebase's other interval-timing call + // sites (LivenessTracker::track(), pollWatchedTargets() below); + // TSC::ticks() itself falls back to OS::nanotime() when the TSC is + // unavailable/disabled, so this is a strict upgrade with no behavior + // change on hosts without a usable timestamp counter. + u64 pass_wall_ticks = 0; + // Genuine in-safepoint cost of this pass, accumulated by + // runPassManualWalk() across every IterateOverReachableObjects/ + // FollowReferences call it makes (root enum, static-field sweep, ordinary + // expansion, rotation re-expansion) - explicitly excluding + // GetObjectsWithTags (not a safepoint call) and every bookkeeping line in + // between. This, not pass_wall_ticks, is what updatePacing()/ + // maybeRevokeBorrowForRootEnumPass() below actually regulate: JFR + // (jdk.ExecuteVMOperation[operation=HeapWalkOperation]) confirmed the two + // can differ substantially - a pass's non-safepoint bookkeeping must not + // be mistaken for pause-time-SLO pressure and throttle the PID controller + // on its behalf. + u64 safepoint_ticks = 0; + + // Every pass is driven by the manual walk (runPassManualWalk() - + // IterateOverReachableObjects for roots, then a batched array-holder + // FollowReferences per BFS level in expandFrontier()), on every + // collector. The walk issues only JVMTI heap calls, which run inside the + // VM_HeapWalkOperation safepoint; JVMTI's iterators apply the active + // collector's own barriers, so the walk is correct on every collector - + // including ZGC, where concurrent relocation would corrupt a raw-oop + // reader. It reads no raw oop. Batching + // one hop per level keeps each FollowReferences bounded, avoiding the + // multi-hundred-ms-to-second STW pauses a whole-graph FollowReferences + // would impose. + bool manual_first_pass = !_search_started; + if (manual_first_pass) { + _search_started = true; + store(_search_start_ns, OS::nanotime()); + } + + // Root/stack-ref enumeration alone never discovers a root's transitive + // children (runPassManualWalk()'s own comment) - there is no "first pass + // walks the whole graph inline" shortcut here, so every pass (first or + // resumed) takes the same expand-frontier shape. Root/stack-ref + // enumeration itself, though, does NOT run on every pass: its fixed + // native dispatch cost is paid in full regardless of budget (see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment), so it is cadence-gated to the + // first pass, a still-truncated retry from last time, or once + // ROOT_ENUM_MIN_INTERVAL_NS has elapsed since it last ran - not every + // pass, unlike expandFrontier()'s cheap incremental work below. + u64 now_ns = OS::nanotime(); + bool run_root_enum = manual_first_pass || _root_enum_truncated_last_time || + (now_ns - _last_root_enum_ns >= ROOT_ENUM_MIN_INTERVAL_NS); + + int frontier_size_before_pass = _frontier != nullptr ? _frontier->size() : 0; + + u64 call_start_ticks = TSC::ticks(); + runPassManualWalk(jvmti, jni, run_root_enum, _first_pass_budget, + _effective_budget, &edges_admitted, &truncated, + &frontier_cap_hit, &safepoint_ticks); + pass_wall_ticks = TSC::ticks() - call_start_ticks; + // TSC::ticks() is monotonic but not necessarily free of measurement noise + // between the outer call_start_ticks snapshot and the several inner + // TSC::ticks() snapshots safepoint_ticks is built from - clamp rather than + // underflow if the accumulated safepoint portion ever reads back larger + // than the whole-call wall time it's a subset of. + u64 non_safepoint_ticks = + pass_wall_ticks > safepoint_ticks ? pass_wall_ticks - safepoint_ticks : 0; + + store(_passes_run, load(_passes_run) + 1); + _last_pass_gc_finish_epoch = gcFinishEpoch(); + store(_last_pass_ns, OS::nanotime()); + if (!run_root_enum) { + // A pass that ran root/stack-ref enumeration spends _first_pass_budget, + // not _effective_budget - its duration is not a signal about the + // per-pass cost updatePacing() is trying to regulate (expandFrontier()'s + // cheap, per-node expansion calls), so feeding it in here would + // throttle _effective_budget down for every one of those unrelated + // later passes based on a single, deliberately oversized outlier. + updatePacing(safepoint_ticks); + } else { + // Excluded from the budget/cadence controller above, but not from the + // borrow ceiling's revocation check (see maybeRevokeBorrowForRootEnumPass()'s + // own comment) - a root-enum pass's in-safepoint cost is real pause time + // and must still be able to revoke a borrowed-budget grant the pacing + // controller would otherwise keep believing is safe. + maybeRevokeBorrowForRootEnumPass(safepoint_ticks); + } + // Search restart (this class's own header comment): accumulate this + // pass's own in-safepoint cost toward the running total restartSearch() + // will spend into _safepoint_pain_budget once the search reaches a terminal state - + // same TSC::ticks_to_millis() conversion updatePacing() already uses for + // its own pass-duration signal. + _search_pain_ms += TSC::ticks_to_millis(safepoint_ticks); + // Independent leaky bucket for the non-safepoint remainder of this pass + // (root/stack-ref enumeration dispatch, frontier-table admission, + // rotation-candidate collection) - see _cpu_pain_budget's own comment + // (referenceChains.h) for why this needs to be tracked separately from + // both _safepoint_pain_budget above and _pause_pid's safepoint_ticks signal. + // Spent every pass, root-enum or not: none of this cost is + // cadence-gated the way root enum's in-safepoint dispatch is. + _cpu_pain_budget.spend(TSC::ticks_to_millis(non_safepoint_ticks)); + + // Design doc's Termination section, decided in priority order: + // 1. Frontier-size cap hit -> abandon immediately, regardless of TTL. + // 2. No pending frontier entries left (this pass wasn't truncated) AND no + // active leak-accumulation watch -> the reachable graph was fully + // explored within the hop cap with nothing left to keep re-checking; + // natural completion (the hop cap alone is a normal boundary, not + // truncation - see heapReferenceCallback()'s own comment). See the + // _watched_leak_klass_count clause's own comment below for why an + // active watch withholds completion here even with an empty pending + // queue. + // 3. TTL exceeded while work is still pending -> abandon. + // 4. Otherwise stay RUNNING - more pending work, no cap hit yet. + // Write the abandon reason (and every other detail field + // buildAbandonedEvent() reads: _passes_run/_last_pass_ns/_search_start_ns + // above, _frontier's size, ...) BEFORE the _search_state transition below, + // and publish that transition with a release store - dump()'s reader side + // (buildAbandonedEvent()/searchState()) pairs it with an acquire load, so + // observing the new _search_state also guarantees every detail field + // written before this release store is visible too, even on a weakly + // ordered CPU (e.g. arm64) where relaxed stores to two different atomics + // carry no such guarantee. + bool has_pending_frontier = truncated; + int frontier_size_after = _frontier->size(); + if (frontier_cap_hit) { + // Frontier table is full -- no new entries can ever be admitted, so + // frontier_size_after can never exceed frontier_size_before_pass again. + // Deferring to the no-progress detector below (as a prior version of + // this branch did) would never actually reach it: this same `if` would + // keep matching every subsequent pass, permanently short-circuiting the + // else-if chain before _passes_since_last_progress is ever read. Abandon + // immediately instead, matching this function's own design-doc priority + // list above (frontier-size cap hit abandons regardless of TTL). + store(_abandon_reason, (u8)SearchAbandonReason::FRONTIER_CAP); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + enqueuePendingAbandonedEvent(); + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass frontier cap hit -- " + "abandoning search (size=%d)", + frontier_size_after); + } else if (!has_pending_frontier && _watched_leak_klass_count == 0) { + storeRelease(_search_state, (u8)SearchState::COMPLETED); + } else if (!has_pending_frontier) { + // Reachable graph fully explored, but LivenessTracker still has at least + // one klass under active leak watch (_watched_leak_klass_count's own + // comment) - do NOT complete. Rotation (collectLeakAccumulationCandidates + // ForRotation() et al., runPassManualWalk()'s own comment) exists + // precisely to re-observe already-EXPANDED entries whose fields mutate + // after their one-time expansion - e.g. an element appended to a + // static-field-rooted collection well after the walk first visited it. + // Once every reachable object has been visited once, has_pending_frontier + // goes permanently false and runPass()'s terminal-state branch would + // otherwise make every future call to this method a no-op forever + // (search_state != RUNNING short-circuits before rotation ever runs + // again) - silently disabling the one mechanism built to catch that + // exact mutation. Falling through here leaves _search_state at RUNNING, + // so the next pass (still gated by shouldRunPass()'s normal cadence/pain + // budget) runs the rotation collectors again with a fresh view of + // whatever object identities LivenessTracker is currently watching. + // _passes_since_last_progress below still counts this pass as "no + // progress" (frontier size is genuinely unchanged), so a watch that + // never resolves anything still bounds out via the TTL/no-progress + // branch below once isUrgent() clears - this only keeps a search alive + // while there is an active signal to keep probing for, not forever + // unconditionally. + } else if (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && + !isUrgent()) { + // The frontier hasn't grown for NO_PROGRESS_PASS_LIMIT consecutive + // passes — the search is genuinely stuck (not just slow), so + // abandon. A large heap takes more passes simply because + // there are more objects to explore; that is not "stuck". + // Only abandon when the frontier stops growing entirely. + // Suppressed when urgent (secondsToOOM() < OOM_URGENT_THRESHOLD_S): + // the search must complete to find the leak before the app OOMs. + store(_abandon_reason, (u8)SearchAbandonReason::TTL); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + enqueuePendingAbandonedEvent(); + } else if (_candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) == + (u64)_candidate_count) { + // Canary early termination: all leaked candidates have been + // found -- the search is complete. + storeRelease(_search_state, (u8)SearchState::COMPLETED); + Counters::increment(REFERENCE_CHAIN_CANDIDATES_FOUND, + __builtin_popcountll(_candidate_found_bits)); + } else if (_candidate_count > 0 && + _passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT && + _passes_since_last_candidate_progress >= + canaryStuckPassLimit()) { + // Canary-specific stuck detector - deliberately NOT suppressed by + // isUrgent() (contrast the ordinary TTL check above). The ordinary + // check's !isUrgent() guard protects a search that's still making real + // (whole-graph) progress from being killed just because the process is + // close to OOM; but _passes_since_last_candidate_progress only advances + // when NO candidate has been newly found and NO new candidate has been + // admitted, which frontier growth elsewhere in the graph does not + // affect. A canary that has made zero discovery progress for this many + // passes is provably not converging regardless of urgency, so letting + // isUrgent() keep it RUNNING would only burn urgency-boosted STW pause + // budget during the same OOM approach this search exists to diagnose. + // + // Also requires the whole-graph frontier to have stalled + // (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT): a search + // whose frontier is still growing is making real progress toward + // eventually reaching the candidate even if it hasn't yet, so it is + // not "stuck" in the sense this detector exists to catch - see + // canaryStuckPassLimit()'s own comment for why the pass limit itself + // also escalates across consecutive restarts of the same chase. + store(_abandon_reason, (u8)SearchAbandonReason::CANARY_STUCK); + storeRelease(_search_state, (u8)SearchState::ABANDONED); + enqueuePendingAbandonedEvent(); + if (_canary_stuck_restart_count < MAX_CANARY_STUCK_BACKOFF_SHIFT) { + _canary_stuck_restart_count++; + } + } + + // Track progress: if the frontier grew this pass, reset the no-progress + // counter. Otherwise increment it. + if (frontier_size_after > frontier_size_before_pass) { + _passes_since_last_progress = 0; + } else { + _passes_since_last_progress++; + } + + // Track canary-specific progress separately - see + // _passes_since_last_candidate_progress's own comment for why frontier + // growth above does not substitute for this. + // Pass-cost EMA for the canary lane's work-scaled backoff (see + // _canary_pass_ema_ns's own comment) - updated from every pass's whole- + // call wall duration so it is warm before the first held-off decision. + // 0.8/0.2, same smoothing as _gotw_ema_call_ns. + { + // Nanosecond EMA: TSC ticks -> ns via the calibrated frequency. A u64 + // ns value does not overflow (u64 ns ~ 584 years). Sub-millisecond + // passes contribute their real cost - see _canary_pass_ema_ns's own + // comment for why integer milliseconds were wrong here. + u64 freq = TSC::frequency(); + u64 pass_wall_ns = freq != 0 ? pass_wall_ticks * 1000000000ULL / freq : 0; + _canary_pass_ema_ns = _canary_pass_ema_ns == 0 + ? pass_wall_ns + : _canary_pass_ema_ns * 4 / 5 + pass_wall_ns / 5; + } + int candidate_progress_mark = + _candidate_count + (int)__builtin_popcountll(_candidate_found_bits); + if (candidate_progress_mark > _last_candidate_progress_mark) { + _last_candidate_progress_mark = candidate_progress_mark; + _passes_since_last_candidate_progress = 0; + // Real chase progress (a candidate found or a new one admitted) - the + // canary lane gets its back-to-back spacing back (multiplier 1, see + // _canary_backoff_mult's own comment). + _canary_backoff_mult = 1; + } else { + _passes_since_last_candidate_progress++; + // No chase progress: double the canary lane's work-scaled spacing + // multiplier, capped. Only while a chase is actually open - an + // all-found or candidate-free pass should not accumulate backoff for + // a chase that no longer exists. + if (_candidate_count > 0 && + __builtin_popcountll(_candidate_found_bits) < (u64)_candidate_count) { + _canary_backoff_mult = + std::min(_canary_backoff_mult * 2, CANARY_BACKOFF_MULT_MAX); + _last_canary_pass_ns = OS::nanotime(); + } + } + + if (load(_search_state) != SearchState::RUNNING) { + _tags_released = releaseSearchTags(jvmti, jni); + // The old canary marker-tag release loop (GetObjectsWithTags per slot, + // then SetTag(0) + DeleteLocalRef per hit) is gone. It was dead AND + // harmful: _candidate_tags[] has been all-zero since marker tags were + // retired ("no marker tags - using leak tags now", pollWatchedTargets), + // and GetObjectsWithTags(1, &tag=0) matches EVERY untagged live object + // per the JVMTI contract - a full-heap enumeration per candidate slot + // on every search termination, each hit also dereferencing a possibly + // null JNIEnv (runPass()'s own early-exit only guards jvmti/_frontier). + // There are no marker tags to release: leak tags belong to + // LivenessTracker's pool and are released by its own cleanup paths. + // Only the bookkeeping reset remains. + if (_candidate_count > 0) { + _candidate_count = 0; + _candidate_found_bits = 0; + memset(_candidate_discovered_count, 0, sizeof(_candidate_discovered_count)); + memset(_candidate_qualifying_tid_count, 0, + sizeof(_candidate_qualifying_tid_count)); + _passes_since_last_candidate_progress = 0; + } + // Only CANARY_STUCK should keep escalating canaryStuckPassLimit() - + // any other terminal reason (natural completion, all candidates found, + // frontier cap, TTL) is an unrelated outcome for this chase sequence, + // so a fresh restart afterward should start back at the base limit. + if (load(_abandon_reason) != SearchAbandonReason::CANARY_STUCK) { + _canary_stuck_restart_count = 0; + } + } + + if (out_truncated != nullptr) { + *out_truncated = truncated; + } + + TEST_LOG_SUMMARY("ReferenceChainTracker::runPass done: edges_admitted=%d truncated=%d " + "frontier_cap_hit=%d searchState=%d abandonReason=%d frontierSize=%d " + "effectiveBudget=%d effectiveCadenceNs=%llu pendingExpand=%zu priorityExpand=%zu " + "candidateFound=%d/%d discoveredCounts=[%d,%d,%d,%d,%d]", + edges_admitted, truncated, frontier_cap_hit, (int)load(_search_state), + (int)_abandon_reason, _frontier->size(), _effective_budget, + (unsigned long long)_effective_cadence_ns, + _pending_expand.size(), _priority_expand.size(), + (int)__builtin_popcountll(_candidate_found_bits), _candidate_count, + _candidate_count > 0 ? _candidate_discovered_count[0] : 0, + _candidate_count > 1 ? _candidate_discovered_count[1] : 0, + _candidate_count > 2 ? _candidate_discovered_count[2] : 0, + _candidate_count > 3 ? _candidate_discovered_count[3] : 0, + _candidate_count > 4 ? _candidate_discovered_count[4] : 0); + + // The pass EXECUTED (the early-exit guards above are the only false + // paths): runPassManualWalk() reports its failures via the + // truncated/frontier_cap_hit out-params, and a frontier-cap hit is a + // legitimate pass outcome (it drives the search's abandon logic), not a + // pass failure - callers treat a false return as "the pass did not run". + // An earlier draft returned `err == JVMTI_ERROR_NONE` where err was + // unconditionally JVMTI_ERROR_NONE (dead logic); the outcome-carrying + // state is the out-params plus the search state, not this bool. + return true; +} + +// --------------------------------------------------------------------------- +// Pause-time-SLO feedback loop (see this method's declaration in +// referenceChains.h for the full mechanism). +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::updatePacing(u64 pass_wall_ticks) { + // Truncating to whole milliseconds matches every other PidController usage + // in this codebase (ObjectSampler/MallocTracer/NativeSocketSampler all feed + // it integer counts, pidController.h's `compute(u64 input, ...)`) - sub-ms + // precision is not meaningful against a millisecond-scale target anyway. + // TSC::ticks_to_millis() already falls back to a nanotime-based conversion + // when the TSC is unavailable/disabled (tsc.h), matching runPass()'s own + // TSC::ticks() fallback for pass_wall_ticks itself. + u64 pass_ms = TSC::ticks_to_millis(pass_wall_ticks); + // time_delta_coefficient is deliberately 1.0, not a real-elapsed-time + // ratio - unlike ObjectSampler's usage (objectSampler.cpp), which + // rescales an event count accumulated over a variable-length real-time + // window against a fixed-real-time target, _pause_pid was constructed + // with sampling_window=1 (its own constructor comment above, in start()): + // one compute() call *is* one pass, and pass_ms already IS the per-call + // quantity being compared against the per-call ceiling _target encodes. + // Rescaling pass_ms by how much real wall-clock time elapsed since the + // previous call would compare it against a target calibrated for a + // different unit (per-second, not per-pass), double-counting the same + // irregular-cadence effect this coefficient exists to correct for in the + // per-second case. (Re-litigated after review: an earlier pass flagged + // this as a bug and a fix using TSC-measured elapsed time was drafted, + // but re-checking against this constructor's own documented design + // confirmed 1.0 is correct here - see this comment instead of changing + // it again.) + double signal = _pause_pid.compute(pass_ms, 1.0); + + // Budget-borrowing (referenceChains.h's _borrowed_budget comment): only a + // sustained run of comfortably-under-target passes earns extra headroom + // above _budget, and any pass that is not comfortably under target revokes + // it immediately - _budget itself must stay the ceiling the instant this + // search stops proving it has pause-time room to spare. + bool comfortably_under_target = + _effective_pause_target_ms > 0 && + (double)pass_ms <= (double)_effective_pause_target_ms * BORROW_UNDER_TARGET_FRACTION; + if (comfortably_under_target) { + if (_consecutive_under_target_passes < BORROW_WARMUP_PASSES) { + _consecutive_under_target_passes++; + } + if (_consecutive_under_target_passes >= BORROW_WARMUP_PASSES) { + int64_t max_borrow = (int64_t)_budget * (BORROW_CEILING_MULTIPLIER - 1); + int64_t grown = _borrowed_budget + + (int64_t)std::llround((double)_budget * BORROW_GROWTH_FRACTION); + _borrowed_budget = std::min(grown, max_borrow); + } + } else { + _consecutive_under_target_passes = 0; + _borrowed_budget = 0; + } + + int64_t ceiling = (int64_t)_budget + _borrowed_budget; + int64_t floor = ceiling > 0 ? std::min((int64_t)MIN_EFFECTIVE_BUDGET, ceiling) + : 0; + int64_t desired = (int64_t)_effective_budget + (int64_t)std::lround(signal); + int64_t clamped = std::max(floor, std::min(ceiling, desired)); + // Whatever part of `desired` the clamp above could not absorb - positive + // when there was more headroom than the ceiling allows, negative when the + // pass is still over the pause-time target even at the floor. Drives + // _effective_cadence_ns below, per this method's own comment on folding + // Open Question 5 into the same controller output. + int64_t overflow = desired - clamped; + _effective_budget = (int)clamped; + + if (overflow < 0) { + // Still over the pause-time ceiling even at the minimum budget - widen + // the fallback interval instead of shrinking the budget further. + u64 step = (u64)(-overflow) * CADENCE_NS_PER_EDGE_OVERFLOW; + _effective_cadence_ns = + std::min(_effective_cadence_ns + step, MAX_EFFECTIVE_CADENCE_NS); + } else if (overflow > 0) { + // Comfortably under the ceiling even at the maximum (config) budget - + // relax the fallback interval. The GC-finish-epoch trigger already fires + // independently of cadence (shouldRunPass() above), so this only + // shortens how long an idle, no-GC-event search waits between passes. + u64 step = (u64)overflow * CADENCE_NS_PER_EDGE_OVERFLOW; + _effective_cadence_ns = + step >= _effective_cadence_ns + ? MIN_EFFECTIVE_CADENCE_NS + : std::max(_effective_cadence_ns - step, MIN_EFFECTIVE_CADENCE_NS); + } + // overflow == 0: the budget clamp alone fully absorbed this pass's + // correction - leave the cadence at its current value. +} + +// A root/stack-ref enumeration pass never reaches updatePacing() above (see +// runPass()'s own comment on why its wall-clock cost is excluded from the +// per-pass PID/effective-budget signal), but it still spends real +// pause-time-SLO time. _borrowed_budget's own comment requires the grant be +// revoked the instant ANY pass is not comfortably under target, so this +// mirrors updatePacing()'s comfortably_under_target check for that one +// purpose only - it never grows _consecutive_under_target_passes/ +// _borrowed_budget, since the warmup streak is calibrated against +// expandFrontier()'s per-node cost, not this call's unrelated fixed +// dispatch cost. +void ReferenceChainTracker::maybeRevokeBorrowForRootEnumPass( + u64 pass_wall_ticks) { + if (_effective_pause_target_ms <= 0) { + return; + } + u64 pass_ms = TSC::ticks_to_millis(pass_wall_ticks); + bool comfortably_under_target = + (double)pass_ms <= (double)_effective_pause_target_ms * BORROW_UNDER_TARGET_FRACTION; + if (!comfortably_under_target) { + _consecutive_under_target_passes = 0; + _borrowed_budget = 0; + // The ceiling updatePacing() would compute right now collapses to + // _budget alone (no _borrowed_budget term above) - re-clamp + // _effective_budget immediately instead of leaving the borrow-inflated + // value in place until the next ordinary pass's updatePacing() call. + _effective_budget = std::min(_effective_budget, (int)_budget); + } +} + +// --------------------------------------------------------------------------- +// Target-selection bridging step - LivenessTracker's leak-candidate ranking feeds +// this tracker's already-running BFS search (design doc's Open Question 3, +// corrected mechanism - see this method's own comment below for why the +// correction replaces the design doc's original seeding proposal). +// --------------------------------------------------------------------------- + +void ReferenceChainTracker::requeueChainRootForRotation(jlong tag) { + if (_frontier == nullptr || tag <= 0) { + return; + } + // Walk the parent chain up to the root-attached entry - the same links + // reconstructChain() walks, but we only need the tag, not the class ids. + // Bounded by _hop_cap (the frontier's own invariant: depth <= hop_cap), + // so a corrupt cycle cannot spin here. + jlong root_tag = tag; + FrontierEntry entry{}; + int hops = 0; + while (hops++ < _hop_cap) { + if (!_frontier->lookup(root_tag, &entry) || entry.parent_tag == 0) { + break; + } + root_tag = entry.parent_tag; + } + if (root_tag == tag) { + return; // tag IS the root - nothing above it to requeue + } + if (!_frontier->lookup(root_tag, &entry) || + entry.state != FrontierEntryState::EXPANDED) { + return; // root pruned or still pending expansion - nothing to re-walk + } + if (isQueuedForRotation(root_tag) || + _priority_expand.size() >= PRIORITY_EXPAND_CAP) { + return; + } + TEST_LOG("ReferenceChainTracker::requeueChainRootForRotation root_tag=%lld " + "target_tag=%lld", + (long long)root_tag, (long long)tag); + _priority_expand.push_back(root_tag); + _priority_expand_set.insert(root_tag); +} + +namespace { + +// The discovered-chain gate's suppression predicate, shared by EVERY site +// that caches a resolved chain - the poll's discovered-instances loop AND +// both representative build paths (the canary/marker path and the +// normal-tag path). Chains shallower than the first real holder hop +// (depth < 2) rooted at a TRANSIENT root (stack local / JNI local) are the +// observed noise shape - a momentarily-live frame's variable holding the +// instance - whose retention explanation evaporates when the frame dies. +// Everything else is real: a depth==1 chain rooted at a durable root is the +// singleton-collection leak shape (a depth-0 static root's elements are +// depth 1), and a depth==0 chain rooted at a durable root is the +// direct-retention shape (the root-retained object itself - e.g. a static +// field's value, or a Thread object for thread-local leaks) - suppressing +// those unconditionally would drop exactly the retention categories the +// search exists to report. Anything deeper passes regardless of root kind +// (at depth >= 2 the chain has at least one real holder hop). +// Representative-driven builds used to bypass this check entirely (found +// live: the canary path cached a stack-local-rooted depth-1 chain for a +// seeded noise-class representative and snapshot-and-keep re-emitted it +// forever) - every cacheResolvedChain() call site in pollWatchedTargets() +// must pass this gate. +bool suppressChainEvent(const ReferenceChainEvent &event) { + return event._depth < 2 && isTransientRootKind(event._root_kind); +} + +} // namespace + +// Chain-event reconstruction for a discovered/correlated instance (out of +// line from referenceChains.h so these sites share the TU's level-gated +// TEST_LOG; per-instance outcomes are level-2 diagnostics). +bool ReferenceChainTracker::buildChainEvent(jvmtiEnv *jvmti, JNIEnv *jni, + jlong target_tag, + ReferenceChainEvent *out) { + if (_frontier == nullptr || out == nullptr) { + TEST_LOG("ReferenceChainTracker::buildChainEvent false: " + "frontier=%p out=%p", (void *)_frontier, (void *)out); + return false; + } + FrontierEntry entry{}; + if (!_frontier->lookup(target_tag, &entry)) { + TEST_LOG("ReferenceChainTracker::buildChainEvent false: " + "target_tag=%lld not in frontier", (long long)target_tag); + return false; + } + std::vector chain; + std::vector edges; + u8 root_kind = 0; + FrontierEntry terminal{}; + if (!_frontier->reconstructChain(target_tag, &chain, &root_kind, &edges, + &terminal)) { + TEST_LOG("ReferenceChainTracker::buildChainEvent false: " + "reconstructChain failed for target_tag=%lld", + (long long)target_tag); + return false; + } + appendStaticFieldRootType(terminal, &chain, &edges); + TEST_LOG("ReferenceChainTracker::buildChainEvent target_tag=%lld chain_size=%zu " + "chain[0]=%u depth=%u root_kind=%u leak_tag=%lld", + (long long)target_tag, chain.size(), chain.empty() ? 0u : chain[0], + entry.depth, (unsigned)root_kind, (long long)entry.leak_tag); + out->_target_tag = entry.leak_tag != 0 ? (u64)entry.leak_tag : (u64)target_tag; + out->_depth = entry.depth; + out->_root_kind = root_kind; + out->_hops.resize(chain.size()); + for (size_t i = 0; i < chain.size(); i++) { + out->_hops[i].klass_id = chain[i]; + } + // Retention-edge labels, aligned with the hops (see fillHopEdgeLabels()). + fillHopEdgeLabels(jvmti, jni, edges, &out->_hops); + return true; +} + +// Appends the root TYPE as a chain element for a static-field-rooted chain: +// the frontier path's root-side end is the static field's HOLDER instance +// (the object stored in the field), but the chain's root is the DECLARING +// CLASS - the holder is "the field instance referenced by the root type", +// one hop below it. Without this element a root-first reading of the chain +// starts at the holder and the root type is only present as the rootKind +// event field and the holder hop's edge label. The declaring class's raw +// class tag is recorded on the root-attached entry at admission time +// (FrontierEntry::referrer_class_tag, captured from referrer_tag_ptr for +// root-attached static edges); it is resolved to a StringDictionary id here +// - this is why the append lives on the tracker, not in FrontierTable. +// Skipped when the root kind is not STATIC_FIELD (thread/JNI roots have no +// further expressible root object - the root-attached entry IS the root +// instance or its nearest class), when no declaring-class tag was captured +// (e.g. heapRootCallback-admitted static roots - the root callback carries +// no referrer_tag_ptr), or when the class tag no longer resolves (class +// unloaded). edges gains one matching entry (the root edge - kind label +// only, the field identity belongs to the holder hop) so the +// _hops[i].edge_label non-empty invariant recordReferenceChain() relies on +// to emit labels at all is preserved. +void ReferenceChainTracker::appendStaticFieldRootType( + const FrontierEntry &terminal, std::vector *chain, + std::vector *edges) { + if (chain == nullptr || + terminal.root_kind != (u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD || + terminal.referrer_class_tag == 0) { + return; + } + u32 root_klass = classTags()->resolve(terminal.referrer_class_tag); + if (root_klass == 0) { + return; + } + chain->push_back(root_klass); + if (edges != nullptr) { + ChainHopEdge root_edge{}; + root_edge.field_index = -1; + root_edge.edge_kind = terminal.root_kind; + root_edge.referrer_class_tag = 0; + edges->push_back(root_edge); + } +} + +// Canary chain reconstruction (out of line for the same reason). The +// canary's build outcomes are level-1: they are the chase's lifecycle +// story, rare (bounded by the candidate count) and chase-relevant. +bool ReferenceChainTracker::buildCanaryChainEvent(int candidate_idx, + ReferenceChainEvent *out) { + if (_frontier == nullptr || out == nullptr || candidate_idx < 0 || + candidate_idx >= _candidate_count) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "frontier=%p out=%p idx=%d count=%d", + (void *)_frontier, (void *)out, candidate_idx, + _candidate_count); + return false; + } + jlong parent_tag = _candidate_parent_tags[candidate_idx]; + u32 candidate_klass = _candidate_referrer_klasses[candidate_idx]; + jlong frontier_tag = _candidate_frontier_tags[candidate_idx]; + std::vector chain; + u8 root_kind = 0; + // The root-attached entry the walk ends at - both branches below leave + // `entry` holding it (the walk's last lookup, or the candidate's own + // entry for a root-referenced candidate). + FrontierEntry terminal{}; + if (parent_tag > 0) { + // Walk parent_tag back to root through the frontier table. + FrontierEntry entry{}; + if (!_frontier->lookup(parent_tag, &entry)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "parent_tag=%lld not in frontier (candidate=%d)", + (long long)parent_tag, candidate_idx); + return false; + } + root_kind = entry.root_kind; + // Hop/cycle bound, mirroring FrontierTable::reconstructChain()'s own + // guard: cycles CAN land here (an entry written before the parent-cycle + // guard existed, or a dangling parent from a concurrent restart), and + // an unbounded `tag = entry.parent_tag` walk on the BFS thread is a + // hang. maxCapacity() bounds the walk to the table's own slot count - + // any legitimate chain is shorter. + int hops = 0; + for (jlong tag = parent_tag; tag > 0 && hops++ <= _frontier->maxCapacity();) { + if (!_frontier->lookup(tag, &entry)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "chain walk: tag=%lld not in frontier (candidate=%d)", + (long long)tag, candidate_idx); + return false; + } + chain.push_back(entry.referrer_klass); + tag = entry.parent_tag; + } + if (hops > _frontier->maxCapacity()) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "chain walk exceeded frontier capacity (candidate=%d) " + "- cycle or dangling parent", + candidate_idx); + return false; + } + terminal = entry; + } else if (parent_tag == 0 && frontier_tag > 0) { + // Root-referenced candidate: chain is just [candidate_klass]. + // root_kind was stored in the frontier entry at pruning time; + // re-read it from the frontier table (the candidate's own entry). + FrontierEntry entry{}; + if (!_frontier->lookup(frontier_tag, &entry)) { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "frontier_tag=%lld not in frontier (candidate=%d)", + (long long)frontier_tag, candidate_idx); + return false; + } + root_kind = entry.root_kind; + terminal = entry; + } else { + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent false: " + "never pruned (candidate=%d parent_tag=%lld frontier_tag=%lld)", + candidate_idx, (long long)parent_tag, + (long long)frontier_tag); + return false; // never pruned (candidate not reached) + } + // Prepend the candidate's own referrer_klass. + chain.push_back(candidate_klass); + // The chain was built root-to-parent; reverse to get candidate-to-root. + std::reverse(chain.begin(), chain.end()); + // Same root-type element buildChainEvent() appends: the canary walk's + // terminal entry is the root-attached entry, and for a static-field root + // the declaring class belongs at the chain's root-side end (after the + // reverse). No-op for other root kinds. + appendStaticFieldRootType(terminal, &chain, nullptr); + out->_target_tag = (u64)frontier_tag; + out->_depth = _candidate_depths[candidate_idx]; + out->_root_kind = root_kind; + const size_t chain_size = chain.size(); + out->_hops.resize(chain_size); + for (size_t i = 0; i < chain.size(); i++) { + out->_hops[i].klass_id = chain[i]; + } + TEST_LOG_SUMMARY("ReferenceChainTracker::buildCanaryChainEvent candidate=%d " + "parent_tag=%lld chain_size=%zu", + candidate_idx, (long long)parent_tag, chain_size); + return true; +} + +void ReferenceChainTracker::pollWatchedTargets(jvmtiEnv *jvmti, JNIEnv *jni) { + if (!_enabled || jvmti == nullptr || jni == nullptr || + !LivenessTracker::instance()->gcGenerationsEnabled()) { + // Explicit guard, even though selectLeakCandidates() below already + // returns 0 candidates whenever its own _gc_generations gate + // (livenessTracker.h) is off - keeps this method's cost at the four + // checks above, not even a shared-lock-guarded table scan, when the + // feature isn't in use (design doc's Open Question 3 "still undecided" + // fallback: referencechains=... alone gets no target-seeding). + return; + } + + // Best-effort drain of any deferred resolved-chain invalidations that a + // pass ending between polls did not already apply (see + // runPassManualWalk()'s end-of-pass drain): chains rebuilt this poll must + // not be gated by a stale cache entry the callback asked to drop. + drainPendingChainInvalidations(); + + // Stamp every entry this poll refreshes with the current search + // generation. _search_start_ns changes each time restartSearch() begins a + // new search (runPass() sets it on the restarted search's first pass); a + // cached chain whose source_search_ns predates the current one was + // reconstructed from a FrontierTable the restart has since reset, so it is + // refreshed below the moment the restarted search re-tags its sample - + // trusting the stale source_tag would risk matching a tag the reset has + // reassigned to an unrelated object. + const u64 current_search_ns = load(_search_start_ns); + + // klass_ids resolved (and therefore already pruned-if-dead) by the + // candidate loop below, so the prune pass afterwards skips re-resolving + // them - it only needs to cover cached klasses that are no longer flagged. + // Per-instance caching: no per-klass prune needed (see comment below + // where the prune logic used to be). + + // Sized generously above LivenessTracker::selectLeakCandidates()'s own + // private MAX_LEAK_CANDIDATES cap (design doc: top 3-5) - that method + // clamps internally to whichever of `max`/its own cap/the qualifying- + // candidate count is smallest, so this local bound only needs to be + // "large enough", not exactly synchronized to a constant this class has + // no visibility into (MAX_LEAK_CANDIDATES is private to LivenessTracker). + constexpr int kMaxWatchedCandidates = 8; + KlassCandidate candidates[kMaxWatchedCandidates]; + int candidate_count = LivenessTracker::instance()->selectLeakCandidates( + candidates, kMaxWatchedCandidates); + + // Publish this poll's qualifying tids as LivenessTracker's watched-admission + // set (see noteSelectedCandidates()'s own comment, livenessTracker.h): + // exactly the (klass, tid) scope tagLeakInstances() tags and this chase + // intercepts gets its allocations admitted at 100% instead of the default + // 10% ratio lottery. Includes the candidate_count == 0 case, which clears + // the set - the boost must track the candidate selection poll by poll. + LivenessTracker::instance()->noteSelectedCandidates(candidates, + candidate_count); + + // Refresh the faster, un-hysteresis-gated klass_id ranking rotation + // priority uses (see _watched_leak_klass_ids' own comment) - but only + // once selectLeakCandidates() above has ALREADY found at least one + // qualifying candidate via its own slower hysteresis gate: this mechanism + // is meant to crank once the trend detector has triggered, not to run the + // ranking independently before that gate has ever fired. Refreshed even + // when candidate_count's specific candidates are unrelated to whichever + // klass ends up ranked highest by generation count - the two lists serve + // different purposes (canary marker output vs. rotation priority) and are + // deliberately not required to agree. + if (candidate_count > 0) { + // Snapshot the OLD watched set before overwriting it, so any klass_id + // that's newly appearing this refresh can get its one-time retroactive + // catch-up (seedLeakAccumulationForNewlyWatchedKlass() - see + // _watched_leak_klass_ids' own comment for why admission-time tracking + // alone cannot see objects admitted before watching started). + u32 previously_watched[MAX_WATCHED_LEAK_KLASSES]; + // Clamp before copying: the count is set by topKlassesByGenerationCount() + // (which caps at MAX), but that is an inter-module invariant with no + // assert at the producing side - a future contract change or a logic + // bug returning raw klass counts would overflow this fixed array here. + int previously_watched_count = + _watched_leak_klass_count > MAX_WATCHED_LEAK_KLASSES + ? MAX_WATCHED_LEAK_KLASSES + : _watched_leak_klass_count; + for (int i = 0; i < previously_watched_count; i++) { + previously_watched[i] = _watched_leak_klass_ids[i]; + } + _watched_leak_klass_count = LivenessTracker::instance()->topKlassesByGenerationCount( + _watched_leak_klass_ids, MAX_WATCHED_LEAK_KLASSES); + for (int i = 0; i < _watched_leak_klass_count; i++) { + u32 klass_id = _watched_leak_klass_ids[i]; + bool already_watched = false; + for (int j = 0; j < previously_watched_count; j++) { + if (previously_watched[j] == klass_id) { + already_watched = true; + break; + } + } + if (!already_watched) { + seedLeakAccumulationForNewlyWatchedKlass(klass_id); + } + } + } + // Only log when there are candidates to act on - this poll runs on every + // BFS-thread wake (once per second), so logging a zero count is per-second + // noise for the common idle case. + if (candidate_count > 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate_count=%d", candidate_count); + // Admit any candidate selectLeakCandidates() returns this poll that + // doesn't already occupy a slot, into the next free slot. This runs on + // every poll (not gated to "only the first time") because + // selectLeakCandidates()'s result set can change across polls - a new + // klass_id can start qualifying well after the search began. Slots are + // never retired or reassigned once occupied: a klass_id that stops + // qualifying just keeps whatever slot it has (and can still be found + // there), it is never freed for reuse by a different klass_id. That + // keeps the marker tag (MARKER_TAG_BASE - slot) a stable, search-lifetime + // identity for heapReferenceCallback()'s decode (referenceChains.cpp, + // near the *tag_ptr <= MARKER_TAG_BASE check) - reusing a slot mid-search + // would let a live marker tag on one object suddenly decode to a + // different klass_id's bookkeeping. + // Use resolveCandidateRepresentative() (re-reads under lock) + // instead of candidates[i].representative (stale jweak). + for (int i = 0; i < candidate_count; i++) { + u32 klass_id = candidates[i].klass_id; + bool already_tracked = false; + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] == klass_id) { + already_tracked = true; + break; + } + } + if (already_tracked) { + continue; + } + if (_candidate_count >= MAX_LEAK_CANDIDATES_FROM_LT) { + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary: klass_id=%u " + "qualifies but all %d slots are occupied - not tracked this search", + klass_id, MAX_LEAK_CANDIDATES_FROM_LT); + continue; + } + // Tag this candidate's specific representative object with a distinct + // marker tag (MARKER_TAG_BASE - slot) so heapReferenceCallback() can + // identify that exact object by identity when the walk reaches it - + // matching by class alone would record a chain for whichever instance + // of that class the walk happens to visit, not necessarily the one + // LivenessTracker flagged as growing. + int slot = _candidate_count; + _candidate_klass_ids[slot] = klass_id; + _candidate_tags[slot] = 0; // no marker tags — using leak tags now + _candidate_count = slot + 1; + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary: admitted klass_id=%u " + "into slot=%d (candidate_count now %d)", + klass_id, slot, _candidate_count); + Counters::increment(REFERENCE_CHAIN_CANDIDATE_COUNT, 1); + } + // Refresh the per-slot qualifying-tid snapshot the walk phases read + // (walkCandidateThreadLocals()): zero every slot first, then fill from + // THIS poll's candidates - a klass whose per-tid trend stopped + // qualifying must stop having its tids walked, exactly like it stops + // consuming pool tags (tagLeakInstances() below keeps the same + // per-poll-candidates scope for the same reason). + memset(_candidate_qualifying_tid_count, 0, + sizeof(_candidate_qualifying_tid_count)); + for (int i = 0; i < candidate_count; i++) { + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != candidates[i].klass_id) { + continue; + } + int n = candidates[i].qualifying_tid_count; + if (n > MAX_CANDIDATE_QUALIFYING_TIDS) { + n = MAX_CANDIDATE_QUALIFYING_TIDS; + } + for (int q = 0; q < n; q++) { + _candidate_qualifying_tids[s][q] = candidates[i].qualifying_tids[q]; + } + _candidate_qualifying_tid_count[s] = n; + break; + } + } + // Tag the tracked instances of THIS poll's candidates with leak tags. + // This replaces the old single-representative marker-tag approach — + // the BFS will find these specific leaking objects by tag, not by + // class match, eliminating noise from unrelated instances of the same + // class. Passing the per-poll candidates (not the ever-occupied + // _candidate_klass_ids slots) is deliberate: the candidates carry the + // qualifying tids selectLeakCandidates() just computed, and a klass + // whose per-tid trend stopped qualifying should stop consuming pool + // tags even though its slot persists (tagLeakInstances()'s own + // comment, livenessTracker.h). + int tagged = LivenessTracker::instance()->tagLeakInstances( + jvmti, candidates, candidate_count); + _leak_tags_assigned = tagged; + _leak_tags_resolved = 0; // reset on each tagging round + TEST_LOG("ReferenceChainTracker::pollWatchedTargets tagLeakInstances tagged=%d", + tagged); + } + + for (int i = 0; i < candidate_count; i++) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u", i, + candidates[i].klass_id); + // Deliberately does NOT resolve candidates[i].representative directly: + // that field is a snapshot taken under selectLeakCandidates()'s own + // shared-lock scan, which can go stale (LRU-evicted and + // DeleteWeakGlobalRef()'d by LivenessTracker's cleanup_table(), running + // concurrently on a different thread) at any point between that call and + // this one - see selectLeakCandidates()'s comment (livenessTracker.h) for + // why resolving it here would be undefined behavior, not just a null + // result. resolveCandidateRepresentative() re-reads the table's current + // value for this klass_id and resolves it atomically under the same + // lock, so it is always safe to call from here. + // Per-instance caching: no per-klass prune needed. + const u32 klass_id = candidates[i].klass_id; + jobject obj = LivenessTracker::instance()->resolveCandidateRepresentative( + jni, klass_id); + if (obj == nullptr) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u " + "representative could not be resolved (died/evicted)", + i, klass_id); + // The representative died, but the canary chain (if the candidate + // was pruned by BFS before the representative died) only needs the + // frontier table — not the live representative. Try to build it + // before erasing the cached chain and skipping this candidate. + bool built_from_canary = false; + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != klass_id) continue; + if ((_candidate_found_bits & (1ULL << s)) && + _candidate_frontier_tags[s] != 0) { + jlong canary_ftag = _candidate_frontier_tags[s]; + _resolved_chains_lock.lock(); + bool need = (_resolved_chains.find(canary_ftag) == _resolved_chains.end()); + _resolved_chains_lock.unlock(); + if (need) { + ReferenceChainEvent event; + built_from_canary = buildCanaryChainEvent(s, &event); + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "buildCanaryChainEvent(dead rep, slot=%d) -> %d", + s, (int)built_from_canary); + if (built_from_canary && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d canary_ftag=%lld " + "klass_id=%u (dead-rep path)", + event._depth, (int)event._root_kind, + (long long)canary_ftag, klass_id); + built_from_canary = false; + invalidateResolvedChain(canary_ftag); + } else if (built_from_canary) { + event._start_time = TSC::ticks(); + cacheResolvedChain(canary_ftag, std::move(event), + current_search_ns); + } + } + } + break; + } + if (!built_from_canary) { + // The representative died. Per-instance caching means we don't erase + // by klass_id — chains for other instances of this class may still + // be valid. The dead representative's chain (if any) will expire + // when the search restarts and the frontier is wiped. + } + continue; // candidate died, or was evicted, since LivenessTracker flagged it + } + + { + jclass obj_klass = jni->GetObjectClass(obj); + char *obj_class_name = nullptr; + if (obj_klass != nullptr && + jvmti->GetClassSignature(obj_klass, &obj_class_name, nullptr) == + JVMTI_ERROR_NONE && + obj_class_name != nullptr) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] " + "klass_id=%u class_name=%s", + i, klass_id, obj_class_name); + jvmti->Deallocate((unsigned char *)obj_class_name); + } + if (obj_klass != nullptr) { + jni->DeleteLocalRef(obj_klass); + } + } + + // Corrected mechanism (a correction to the design doc's + // original proposal): a READ, never a SetTag + // seed. runPass()'s whole-graph walk is the only thing that ever + // assigns a tag; if it already has (tag > 0), heapReferenceCallback() + // already recorded a correct parent_tag/depth chain for this object the + // moment it was first visited - pre-tagging it here instead would make + // that callback's `*tag_ptr == 0` branch (the only branch that records + // parent_tag/depth, referenceChains.h) skip it entirely the next time a + // pass reached it. + jlong tag = getTag(jvmti, obj); + + // Canary search: if this candidate was pre-tagged with a marker tag + // (negative, set above), use the canary chain reconstruction. The + // marker tag itself stays on the representative for the whole search + // (heapReferenceCallback() never overwrites it), so this stays true + // regardless of whether the walk has actually reached it yet this pass - + // buildCanaryChainEvent() below is what distinguishes "found" (parent_tag + // or frontier_tag populated) from "not yet pruned". + if (tag <= MARKER_TAG_BASE) { + // The marker tag encodes the slot this object was pre-tagged at + // (MARKER_TAG_BASE - slot, mirroring heapReferenceCallback()'s own + // decode at its MARKER_TAG_BASE check). Decode it from the tag itself + // rather than reusing the loop index `i`: selectLeakCandidates() is + // not guaranteed to return candidates in the same order across polls, + // so `i` can drift from the slot this object was actually tagged at. + int candidate_slot = (int)(MARKER_TAG_BASE - tag); + if (candidate_slot < 0 || candidate_slot >= _candidate_count) { + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary candidate[%d] " + "klass_id=%u marker_tag=%lld decodes to out-of-range slot=%d " + "(candidate_count=%d) - skipping", + i, klass_id, (long long)tag, candidate_slot, _candidate_count); + jni->DeleteLocalRef(obj); + continue; + } + bool need_refresh = false; + jlong canary_ftag = _candidate_frontier_tags[candidate_slot]; + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(canary_ftag); + need_refresh = (it == _resolved_chains.end() || + it->second.source_search_ns != current_search_ns); + _resolved_chains_lock.unlock(); + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary candidate[%d] " + "klass_id=%u marker_tag=%lld slot=%d needRefresh=%d", + i, klass_id, (long long)tag, candidate_slot, need_refresh); + if (need_refresh) { + ReferenceChainEvent event; + bool built = buildCanaryChainEvent(candidate_slot, &event); + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary " + "buildCanaryChainEvent(slot=%d) -> %d", + candidate_slot, built); + if (built && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d canary_ftag=%lld " + "klass_id=%u (canary path)", + event._depth, (int)event._root_kind, + (long long)canary_ftag, klass_id); + built = false; + invalidateResolvedChain(canary_ftag); + } + if (built) { + event._start_time = TSC::ticks(); + cacheResolvedChain(canary_ftag, std::move(event), + current_search_ns); + } + } + // Fall through to discovered-instances check below — the canary + // representative may not have been reached by BFS yet, but other + // instances of the same class may have been admitted and their + // chains can be built now. + } + + // Normal (non-canary) path: tag > 0 means the walk visited this + // object and assigned it a frontier tag. + + // Keep the holder chain's root warm in the rotation queue: a growing + // container's current internals are only reachable via the holder's + // re-walk (requeueChainRootForRotation()'s own comment). Every poll, + // not just on cache refresh - the holder must be re-walked CONTINUOUSLY + // to observe each resize as it happens. + if (tag > 0) { + requeueChainRootForRotation(tag); + } + + // Reconstruct only when this klass has no current chain cached: either + // nothing cached yet, or what is cached was built from a different tag or + // an earlier search generation (see current_search_ns above). A klass + // that keeps getting flagged, unchanged, across many polls is left alone + // - its cached chain is already being re-emitted on every dump. + // + // Round-19 (pod 289f8): resolve by the SLOT's chain key, not the + // representative's leak tag. The resolved-chains cache is keyed by + // frontier tags (disc_tag); under the leak-tag design the rep's leak + // tag is never a frontier key (interceptions insert with a fresh + // frontier tag, entry.leak_tag rides inside), so find(leak_tag) can + // never hit — the rep stayed needRefresh=1 forever, retrying a + // buildChainEvent(leak_tag) that always fails "not in frontier". + // buildDiscoveredInstanceChains() records the slot's frontier tag + // when a leak-tag chain first lands (the found criterion); until + // then there is nothing to refresh — the discovered path below both + // creates the chain and marks the slot found. + bool need_refresh = false; + jlong rep_chain_key = 0; + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] == klass_id) { + rep_chain_key = _candidate_frontier_tags[s]; + break; + } + } + if (rep_chain_key != 0) { + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(rep_chain_key); + need_refresh = (it == _resolved_chains.end() || + it->second.source_search_ns != current_search_ns); + _resolved_chains_lock.unlock(); + } + TEST_LOG("ReferenceChainTracker::pollWatchedTargets candidate[%d] klass_id=%u tag=%lld " + "needRefresh=%d", + i, klass_id, (long long)tag, need_refresh); + if (need_refresh) { + ReferenceChainEvent event; + bool built = buildChainEvent(jvmti, jni, tag, &event); + TEST_LOG("ReferenceChainTracker::pollWatchedTargets buildChainEvent(tag=%lld) -> %d", + (long long)tag, built); + if (built && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d rep_tag=%lld klass_id=%u " + "(representative path)", + event._depth, (int)event._root_kind, (long long)tag, + klass_id); + built = false; + invalidateResolvedChain(tag); + } + if (built) { + // Provisional stamp; drainPendingChainEvents() re-stamps each copy at + // dump time so the event lands in that chunk's window. + event._start_time = TSC::ticks(); + cacheResolvedChain(tag, std::move(event), current_search_ns); + } + } + // tag == 0: The representative object has no tag — the BFS walk + // hasn't reached it yet AND it is not yet leak-tagged. No action here: + // tagLeakInstances() (earlier in this same poll) tags the tracked + // instances of candidate classes' QUALIFYING tids from the reusable + // pool, so the next tagLeakInstances round or the next walk pass will + // pick it up (a representative of a klass whose candidates all carry + // the rep's tid - foldKlassCountsLocked()'s dominant-tid rep bias - is + // in scope by construction). The + // old marker-tag re-tag path is gone — marker tags are no longer the + // candidate discovery mechanism (leak tags are), and re-tagging with + // a marker tag here would resurrect the dead mechanism on objects the + // leak-tag pool has not yet reached. + + // Build chain events for auto-marked discovered instances of this class. + // Runs via buildDiscoveredInstanceChains() - see that method's own + // comment for why it is slot-driven (the orphan fix), and the + // per-poll-candidate call plus the slot sweep below for the two + // call sites. + buildDiscoveredInstanceChains(jvmti, jni, klass_id, current_search_ns); + + jni->DeleteLocalRef(obj); + } + + // Orphan fix: slots whose klass is NOT among this poll's candidates. A + // candidate that qualified long enough for the walk to record discovered + // instances, then stops qualifying (per-tid trend decay, thread switch - + // the exact shape observed live: the walk recorded 8 discovered + // instances the pass AFTER the candidate's seeded trend aged out, and + // nothing ever built their chains for the remaining 116 passes, because + // the per-candidate loop above only iterates the CURRENT poll's + // candidates), must not strand those instances: the slot persists by + // design precisely so the klass "can still be found there" + // (_candidate_klass_ids' own comment above) - this sweep is what finds + // them. Slots whose klass IS a poll candidate are skipped here to avoid + // double-processing (the loop above already handled them) - the build + // itself is idempotent either way (already-cached chains are skipped + // under the same source_search_ns check), this is purely to keep the + // per-poll log volume unchanged for still-qualifying candidates. + for (int s = 0; s < _candidate_count; s++) { + u32 slot_klass = _candidate_klass_ids[s]; + bool in_poll = false; + for (int i = 0; i < candidate_count; i++) { + if (candidates[i].klass_id == slot_klass) { + in_poll = true; + break; + } + } + if (!in_poll && slot_klass != 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets orphan slot " + "sweep: slot=%d klass_id=%u not in poll candidates - building " + "its discovered chains", + s, slot_klass); + buildDiscoveredInstanceChains(jvmti, jni, slot_klass, current_search_ns); + } + } + + // Per-instance caching: chains are keyed by frontier tag, not klass_id. + // There is no per-klass prune — chains for dead instances are harmless + // (they describe a reference path that was valid at resolution time) and + // expire naturally when the search restarts (frontier is wiped, all tags + // become invalid, _resolved_chains is cleared in restartSearch()). + // The backend can filter stale chains by cross-referencing with + // HeapLiveObject events from the same chunk. +} + +// Inserts or refreshes klass_id's resolved chain - see _resolved_chains' +// comment (referenceChains.h) for why a resolved chain is cached and +// re-emitted rather than emitted once. A refresh (klass_id already present) +// always succeeds; only a brand-new klass_id arriving with the cache already +// full is dropped (counted, not silent), rather than evicting some other +// still-live sample's chain. Split out of pollWatchedTargets() so +// ResolvedChainCacheTest (referenceChains_ut.cpp) can drive the overflow path +// directly, without standing up hundreds of real LivenessTracker candidates. +// Returns false when the chain was dropped (cache full for a new source +// tag) so the caller can skip coverage accounting for a chain that will +// never be emitted. +bool ReferenceChainTracker::cacheResolvedChain(jlong source_tag, + ReferenceChainEvent &&event, + u64 source_search_ns) { + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(source_tag); + if (it == _resolved_chains.end() && + (int)_resolved_chains.size() >= MAX_RESOLVED_CHAINS) { + _resolved_chains_lock.unlock(); + Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); + TEST_LOG("ReferenceChainTracker::cacheResolvedChain dropped new source_tag=%lld, " + "cache full (at MAX_RESOLVED_CHAINS=%d)", + (long long)source_tag, MAX_RESOLVED_CHAINS); + return false; + } + CachedChain &slot = _resolved_chains[source_tag]; + slot.event = std::move(event); + slot.source_tag = source_tag; + slot.source_search_ns = source_search_ns; + TEST_LOG("ReferenceChainTracker::cacheResolvedChain source_tag=%lld cache_size=%d", + (long long)source_tag, (int)_resolved_chains.size()); + _resolved_chains_lock.unlock(); + return true; +} + +void ReferenceChainTracker::invalidateResolvedChain(jlong source_tag) { + _resolved_chains_lock.lock(); + auto it = _resolved_chains.find(source_tag); + if (it != _resolved_chains.end()) { + _resolved_chains.erase(it); + TEST_LOG("ReferenceChainTracker::invalidateResolvedChain source_tag=%lld", + (long long)source_tag); + } + _resolved_chains_lock.unlock(); +} + +void ReferenceChainTracker::deferResolvedChainInvalidation(jlong source_tag) { + // Never takes _resolved_chains_lock: called from inside the + // FollowReferences STW pause, where blocking on the JFR dump thread's + // full-cache copy would extend the pause. See the declaration comment. + _pending_chain_invalidations_lock.lock(); + if ((int)_pending_chain_invalidations.size() < + MAX_PENDING_CHAIN_INVALIDATIONS) { + _pending_chain_invalidations.push_back(source_tag); + } + // Overflow drops the deferral: the stale cached chain then survives until + // the search restart wipes _resolved_chains - the same lifetime bound the + // cache itself has. + _pending_chain_invalidations_lock.unlock(); +} + +void ReferenceChainTracker::drainPendingChainInvalidations() { + std::vector pending; + { + _pending_chain_invalidations_lock.lock(); + pending.swap(_pending_chain_invalidations); + _pending_chain_invalidations_lock.unlock(); + } + for (size_t i = 0; i < pending.size(); i++) { + invalidateResolvedChain(pending[i]); + } +} + +// Builds and caches chain events for every auto-marked discovered +// instance recorded against a slot holding klass_id (see the auto-mark +// block in heapReferenceCallback() for how instances get recorded). +// +// Slot-driven, deliberately NOT poll-candidate-driven: the slots persist +// for the whole search ("a klass_id that stops qualifying just keeps +// whatever slot it has ... it is never freed for reuse", +// pollWatchedTargets()'s own slot-admission comment) specifically so the +// klass can still be found after it stops qualifying - a candidate that +// qualified long enough for the walk to record discovered instances and +// then stopped qualifying (per-tid trend decay, thread switch) must not +// strand them. Two call sites make that true: the per-poll-candidate loop +// (klass still qualifying - chains build as fresh as possible, before the +// trend can age out) and the orphan slot sweep (klass no longer in this +// poll's candidates - the recorded instances still get their chains). +// Idempotent per instance: buildDiscoveredInstanceChains skips any tag +// already cached for the current search generation, so calling both +// paths for the same slot in one poll is safe - the sweep simply never +// overlaps the per-candidate call for the same klass. +void ReferenceChainTracker::buildDiscoveredInstanceChains(jvmtiEnv *jvmti, + JNIEnv *jni, + u32 klass_id, + u64 current_search_ns) { +for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != klass_id) continue; + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "discovered loop: klass_id=%u slot=%d discovered_count=%d", + klass_id, s, _candidate_discovered_count[s]); + if (_candidate_discovered_count[s] == 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "no discovered instances for klass_id=%u slot=%d", + klass_id, s); + } + for (int d = 0; d < _candidate_discovered_count[s]; d++) { + jlong disc_tag = _candidate_discovered_tags[s][d]; + if (disc_tag == 0) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "disc_tag=0 at idx=%d for klass_id=%u slot=%d", + d, klass_id, s); + continue; + } + // Skip if already cached for this instance - but only for the CURRENT + // search generation: restartSearch() resets the frontier tag namespace, + // so a cached entry under the same numeric tag from an earlier search + // describes a different object and must not suppress the rebuild (the + // generation check mirrors the rep-refresh paths in pollWatchedTargets()). + const u64 current_search_ns = load(_search_start_ns); + _resolved_chains_lock.lock(); + auto cached_it = _resolved_chains.find(disc_tag); + bool already_cached = (cached_it != _resolved_chains.end() && + cached_it->second.source_search_ns == + current_search_ns); + _resolved_chains_lock.unlock(); + if (already_cached) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "already_cached disc_tag=%lld klass_id=%u slot=%d idx=%d", + (long long)disc_tag, klass_id, s, d); + continue; + } + ReferenceChainEvent event; + bool built = buildChainEvent(jvmti, jni, disc_tag, &event); + // Retention-explanation filter. Only applies to discovered + // instances, not canary. Suppress: + // - depth==0: the instance IS the root (chain is just [object], + // no holder to explain anything); + // - depth==1 rooted at a TRANSIENT root (stack local / JNI + // local): the observed noise shape - a momentarily-live + // frame's variable holding the instance. The chain explains a + // retention that evaporates when the frame dies. + // Both durable-rooted shapes are REAL direct-retention chains and + // must NOT be caught by a blanket depth filter: depth==1 rooted at + // a static field is the singleton-collection leak shape (a depth-0 + // static root's elements are depth 1), and depth==0 rooted at a + // durable root is the root-retained object itself (a static field's + // value, a Thread object for thread-local leaks) - the actual + // retention categories the search exists to report. Anything + // deeper passes regardless of root kind (at depth >= 2 the chain + // has at least one real holder hop). + if (built && suppressChainEvent(event)) { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "filtered depth=%u root_kind=%d disc_tag=%lld klass_id=%u", + event._depth, (int)event._root_kind, + (long long)disc_tag, klass_id); + built = false; + // Also drop any chain cached for this tag before the filter + // existed (or before an improveChain/reparent upgraded it) - + // drainPendingChainEvents() re-emits cached chains + // unconditionally, so suppressing only the build would leave the + // noise chains re-emitting forever. + invalidateResolvedChain(disc_tag); + } + if (built) { + event._start_time = TSC::ticks(); + // Coverage accounting below must only advance for a chain that was + // actually stored - a cache-full drop would let the search report + // the candidate as found without ever emitting its chain. + if (cacheResolvedChain(disc_tag, std::move(event), + current_search_ns)) { + // Track coverage for adaptive CPU budget + if (event._target_tag >= (u64)LEAK_TAG_BASE) { + _leak_tags_resolved++; + // Round-19 (pod 289f8): the canary chase's found criterion. The + // marker-tag design this code replaced ("no marker tags — using + // leak tags now", the slot registration in pollWatchedTargets()) + // never migrated heapReferenceCallback()'s marker-keyed + // found-bit setting — under leak tags nothing ever set + // _candidate_found_bits, so the chase could never exit (0/1 + // across every search, exits only via frontier-cap/no-progress). + // A leak-tag-target chain for the slot IS the leak-tag-world + // "canary found": a walk reached the leaked population and the + // correlation carried (target_tag = the leak tag). Record the + // link so buildCanaryChainEvent()'s root-referenced branch works + // too (parent_tag 0, frontier_tag = disc_tag). + if (!(_candidate_found_bits & (1ULL << s))) { + _candidate_found_bits |= (1ULL << s); + _candidate_frontier_tags[s] = disc_tag; + _candidate_parent_tags[s] = 0; + _candidate_depths[s] = event._depth; + _candidate_referrer_klasses[s] = klass_id; + TEST_LOG_SUMMARY("ReferenceChainTracker::pollWatchedTargets canary " + "found: klass_id=%u slot=%d leak chain target_tag=%llu " + "via disc_tag=%lld (%d/%d candidates found)", + klass_id, s, (unsigned long long)event._target_tag, + (long long)disc_tag, + __builtin_popcountll(_candidate_found_bits), + _candidate_count); + } + } + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "auto-marked chain for klass_id=%u tag=%lld target_tag=%llu", + klass_id, (long long)disc_tag, + (unsigned long long)event._target_tag); + } + } else { + TEST_LOG("ReferenceChainTracker::pollWatchedTargets " + "buildChainEvent failed for discovered tag=%lld " + "klass_id=%u slot=%d disc_idx=%d", + (long long)disc_tag, klass_id, s, d); + } + } + break; +} +} + +void ReferenceChainTracker::recordDiscoveredInstance(u32 klass_id, + jlong frontier_tag, + bool leak_correlated) { + // See the declaration's own comment (referenceChains.h) for the + // noise-eviction rationale. Bounded: at most MAX_DISCOVERED_INSTANCES- + // PER_CLASS frontier lookups when evicting, zero allocation (slots are + // fixed arrays). + for (int s = 0; s < _candidate_count; s++) { + if (_candidate_klass_ids[s] != klass_id) { + continue; + } + if (_candidate_discovered_count[s] < MAX_DISCOVERED_INSTANCES_PER_CLASS) { + _candidate_discovered_tags[s][_candidate_discovered_count[s]++] = + frontier_tag; + TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance slot=%d " + "klass_id=%u tag=%lld leak_correlated=%d count=%d", + s, klass_id, (long long)frontier_tag, (int)leak_correlated, + _candidate_discovered_count[s]); + return; + } + if (!leak_correlated) { + return; // full - noise never displaces anything + } + // All slots full and this instance is leak-correlated: evict the first + // slot held by an entry with no leak tag (a noise instance). Also drop + // the evicted instance's cached chain so it stops re-emitting - the + // discovered-loop gate below suppresses new noise builds, but a chain + // cached before that gate keeps draining forever. + for (int d = 0; d < _candidate_discovered_count[s]; d++) { + jlong victim = _candidate_discovered_tags[s][d]; + FrontierEntry victim_entry{}; + if (_frontier == nullptr || + !_frontier->lookup(victim, &victim_entry) || + victim_entry.leak_tag == 0) { + _candidate_discovered_tags[s][d] = frontier_tag; + // Deferred (never take _resolved_chains_lock inside the STW walk + // callback): drained by drainPendingChainInvalidations() after the + // pass. The stale cached chain may be re-emitted once more before + // the drain - timing-only effect, never correctness. + deferResolvedChainInvalidation(victim); + TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance evicted " + "noise slot=%d idx=%d victim_tag=%lld for leak tag=%lld", + s, d, (long long)victim, (long long)frontier_tag); + return; + } + } + TEST_LOG("ReferenceChainTracker::recordDiscoveredInstance all slots " + "leak-correlated, dropping tag=%lld klass_id=%u", + (long long)frontier_tag, klass_id); + return; + } +} + +bool ReferenceChainTracker::correlateAdmittedLeakTag(jlong frontier_tag, + jlong leak_tag, + u32 klass_id) { + // See the declaration's own comment (referenceChains.h). Called from + // LivenessTracker::tagLeakInstances() on this same thread + // (pollWatchedTargets -> tagLeakInstances), so _candidate_* slot access + // here never races heapReferenceCallback's auto-mark path. + if (_frontier == nullptr) { + return false; + } + FrontierEntry entry{}; + if (!_frontier->lookup(frontier_tag, &entry)) { + return false; // not a live frontier tag (or the search restarted) + } + if (entry.leak_tag != 0) { + // Already correlated (idempotent) - e.g. a second tagLeakInstances + // round after a post-restart re-admission. + return true; + } + _frontier->setLeakTag(frontier_tag, leak_tag); + TEST_LOG_SUMMARY("ReferenceChainTracker::correlateAdmittedLeakTag " + "frontier_tag=%lld leak_tag=%lld depth=%u parent_tag=%lld", + (long long)frontier_tag, (long long)leak_tag, entry.depth, + (long long)entry.parent_tag); + recordDiscoveredInstance(klass_id, frontier_tag, true); + return true; +} + +void ReferenceChainTracker::drainPendingChainEvents( + std::vector *out) { + if (out == nullptr) { + return; + } + // Snapshot-and-keep, not a drain: every cached chain is copied out (and + // re-stamped so it lands in the dumping chunk's window) while the cache + // itself is left intact, so the same live sample's chain re-emits into + // every chunk it survives into (see _resolved_chains' comment). `now` is + // read once, before the lock, so every event in one dump shares a stamp. + u64 now = TSC::ticks(); + _resolved_chains_lock.lock(); + for (const auto &kv : _resolved_chains) { + out->push_back(kv.second.event); + out->back()._start_time = now; + } + _resolved_chains_lock.unlock(); + TEST_LOG_SUMMARY("ReferenceChainTracker::drainPendingChainEvents re-emitted=%d", + (int)out->size()); +} + +void ReferenceChainTracker::enqueuePendingAbandonedEvent() { + // Called right after runPass() (referenceChains.cpp) writes + // SearchState::ABANDONED, on the same thread, before shouldRunPass() gets + // a chance to call restartSearch() - so buildAbandonedEvent()'s live read + // of _search_state/_abandon_reason/etc. is guaranteed to still succeed + // here even though it cannot be trusted to succeed later, from dump()'s + // independent clock (see _pending_abandoned_events' own comment). + ReferenceChainAbandonedEvent event; + if (!buildAbandonedEvent(&event)) { + return; + } + // Stamp when the search actually stopped, not when a later dump writes the + // queued event - an abandon is a point-in-time occurrence and dump() can + // lag it by a whole chunk rotation. + event._start_time = TSC::ticks(); + _pending_abandoned_events_lock.lock(); + if ((int)_pending_abandoned_events.size() >= MAX_PENDING_ABANDONED_EVENTS) { + _pending_abandoned_events_lock.unlock(); + Counters::increment(REFERENCE_CHAIN_EVENTS_DROPPED); + TEST_LOG_SUMMARY("ReferenceChainTracker::enqueuePendingAbandonedEvent dropped, " + "queue full (at MAX_PENDING_ABANDONED_EVENTS=%d)", + MAX_PENDING_ABANDONED_EVENTS); + return; + } + _pending_abandoned_events.push_back(event); + TEST_LOG_SUMMARY("ReferenceChainTracker::enqueuePendingAbandonedEvent reason=%d " + "queue_size=%d", + (int)event._reason, (int)_pending_abandoned_events.size()); + _pending_abandoned_events_lock.unlock(); +} + +void ReferenceChainTracker::drainPendingAbandonedEvents( + std::vector *out) { + if (out == nullptr) { + return; + } + // True drain, unlike drainPendingChainEvents() above: each queued event + // describes a discrete past occurrence, not an ongoing live sample, so + // once Profiler::dump() (profiler.cpp) has emitted it there is nothing + // left to re-report on the next dump. + _pending_abandoned_events_lock.lock(); + out->insert(out->end(), _pending_abandoned_events.begin(), + _pending_abandoned_events.end()); + _pending_abandoned_events.clear(); + _pending_abandoned_events_lock.unlock(); + TEST_LOG_SUMMARY("ReferenceChainTracker::drainPendingAbandonedEvents drained=%d", + (int)out->size()); +} diff --git a/ddprof-lib/src/main/cpp/referenceChains.h b/ddprof-lib/src/main/cpp/referenceChains.h index fb483dd810..1c4c9515a9 100644 --- a/ddprof-lib/src/main/cpp/referenceChains.h +++ b/ddprof-lib/src/main/cpp/referenceChains.h @@ -28,209 +28,1102 @@ #include #include -#include "referenceChainFrontier.h" +// Incremental resumption across passes (see ReferenceChainTracker::runPass() +// below), building on an earlier +// proof-of-concept that established two things end-to-end: +// 1. A cheap "a GC just happened" signal reaches this subsystem via the +// GarbageCollectionStart/Finish JVMTI callbacks (vmEntry.cpp), mirroring +// LivenessTracker::onGC() - just bumping an +// atomic epoch counter, nothing else. +// 2. JVMTI object tags round-trip a live object across a GC (SetTag/GetTag), +// via the minimal tagObject()/getTag()/clearTag() helpers below. +// The tag-indexed FrontierTable was added next, followed by the actual heap +// walk (runPass() calling jvmtiEnv::IterateOverReachableObjects to enumerate +// heap roots via heapRootCallback()/stackRefCallback(), populating +// FrontierTable subject to the hop cap/budget/frontier cap) - but that walk +// originally ran as a single, +// non-resumable pass with no cross-pass persistence, no GC-epoch-driven +// scheduling, and no tag release. This revision makes the search resumable +// and terminating: +// - runPass() now distinguishes a search's first pass (IterateOverReachableObjects +// to enumerate heap roots, exactly as the original single-pass walk did) +// from a resumed pass (expandFrontier() below: resolve each +// not-yet-expanded frontier entry via GetObjectsWithTags - dead ones are +// pruned for free - then call FollowReferences with that object as +// initial_object to discover its own outgoing edges, continuing until +// the per-pass budget or the frontier cap is hit). +// - The Termination section's cutoffs are enforced across passes: the hop +// cap already carried over via FrontierEntry::depth; this adds a +// wall-clock TTL cutoff (_ttl_ms, from first pass) and treats the +// frontier-size cap as immediate search abandonment rather than a +// per-pass truncation. +// - releaseSearchTags() clears (SetTag(obj, 0)) every live tag this search +// still owns once it completes or is abandoned, without discarding the +// FrontierTable's own records - reconstructChain() keeps working from +// memory after the search ends, only the underlying JVMTI tag map entry +// is released (design doc's Open Question 4 concern about leftover-tag +// overhead). +// - shouldRunPass()/threadLoop() implement the Triggering section's pass- +// scheduling signal (GC-finish epoch advanced, or a fixed cadence +// elapsed) - see threadLoop()'s own comment for why the thread this runs +// on is still not spawned by start(). +// +// pollWatchedTargets() below is the LivenessTracker-to- +// ReferenceChainTracker target-selection bridge, closing the gap left by +// buildChainEvent() having no caller. It polls +// LivenessTracker::selectLeakCandidates() (livenessTracker.h's Open Question +// 3 population-slope ranking) and, for each candidate already tagged by an +// ordinary runPass() walk, reconstructs and emits its chain through the +// profiler's dump-time write path. This is a READ of getTag(), never a +// SetTag seed - see pollWatchedTargets()'s own comment for why seeding a +// candidate before the forward walk reached it would break the walk (the +// design doc's original proposal, corrected here). +// +// `can_tag_objects` and `can_generate_garbage_collection_events` are already +// requested unconditionally in vmEntry.cpp, so this bridging step only adds +// callback wiring and lazy event enablement, not capability requests. +// +// The pause-time pacing controller replaces the fixed _budget/PASS_CADENCE_NS +// constants' role as the literal per-pass values with a measured +// pause-time-SLO feedback loop (design doc's Open Questions 2/5, "Proposed +// mechanism" paragraphs). runPass() now times its own FollowReferences/ +// GetObjectsWithTags call (already the thread blocked inside the safepoint +// those trigger, see the Triggering section) and feeds that duration to +// updatePacing() below, which scales _effective_budget/_effective_cadence_ns +// - the values runPass()/shouldRunPass()/threadLoop() now actually use - +// via this tracker's own PidController instance (_pause_pid). _budget/ +// PASS_CADENCE_NS survive as this controller's ceiling/baseline +// respectively, not as the literal per-pass values anymore. See +// updatePacing()'s own comment for the full mechanism, including why its +// gains are not copied from ObjectSampler/MallocTracer/NativeSocketSampler's +// shared triple. +// +// Search restart. Earlier revisions of this class only ever ran a +// single search for the tracker's entire lifetime (runPass()'s own comment +// used to read "starting a *new* search once one ends is not implemented"). That is +// a real gap: LivenessTracker::selectLeakCandidates() only trusts a klass's +// population trend once it has accumulated +// LivenessTracker::KLASS_POPULATION_MIN_FILL_FOR_TREND GC epochs of history +// (livenessTracker.cpp), which takes real wall-clock time - but a +// large-enough-budget search can finish walking the whole reachable graph, +// and permanently stop, before that time has passed. Any object allocated +// after the search already completed is then structurally undiscoverable +// forever, not just unlucky. +// +// The fix: a (re)started search's first pass is now gated on +// LivenessTracker already reporting at least one leak candidate, rather than +// starting unconditionally - by the time a candidate is flagged, the +// underlying object has necessarily survived several epochs already, so a +// fresh root-seeded walk started right then is very likely to still find it +// reachable. restartSearch() (referenceChains.cpp) resets the per-search +// state (frontier table, tag counter, emitted-target set) once a prior +// search reaches COMPLETED/ABANDONED, so shouldRunPass() can treat the next +// candidate-driven trigger exactly like a first-ever search. +// +// Restarting is still an expensive full-heap walk, so canAffordNewSearch() +// also gates it on _safepoint_pain_budget (PainBudget, painBudget.h) - a leaky bucket +// over the wall-clock cost of past searches, not a fixed cooldown, so a +// search that finished cheaply can restart again soon while an expensive one +// has to wait proportionally longer. shouldRunPass() reuses this same +// canAffordNewSearch() check for the very first search too, though the pain +// budget half of it is always a no-op there (nothing has been spent yet). +// With LivenessTracker::gcGenerationsEnabled() off there is no candidate +// signal to gate on at all, so both the first search and every restart start +// unconditionally, exactly as before this revision - gating without a +// leak-detection mechanism running would have no signal to justify one. +// +// JVMTI spec restriction: GarbageCollectionStart/Finish run while the VM is +// at a safepoint, and only the Memory Management category (Allocate/ +// Deallocate) is allowed from inside them - Heap category calls (SetTag, +// GetTag, GetObjectsWithTags, FollowReferences, IterateThroughHeap) are not. +// onGCStart()/onGCFinish() below must therefore never call anything but the +// atomic counter bump. GCCallbackGuard (referenceChains.cpp) marks this +// thread as "inside the GC callback" for the duration of that bump; the tag +// helpers assert() (debug builds only) that they are never entered while the +// guard is active, as a self-consistency check - it does not catch every way +// this restriction could be violated, only calls routed through this class. +// +// Per-tag frontier metadata state (design doc: Frontier/EdgeStore records). +// FRONTIER->EXPANDED is driven by ReferenceChainTracker::expandFrontier()/ +// markAllFrontierExpanded() once an entry's own outgoing edges have +// been visited; FRONTIER/EXPANDED->ABANDONED is driven by expandFrontier()'s +// resolve-or-drop path (dead objects) and releaseSearchTags() (search +// completion/abandonment). +namespace FrontierEntryState { +constexpr u8 FRONTIER = 0; // discovered, not yet expanded by FollowReferences +constexpr u8 EXPANDED = 1; // expanded; children (if any) are in the table +constexpr u8 EDGE = 2; // on a path toward a target sample (EdgeStore) +constexpr u8 ABANDONED = 3; // tag released; entry kept only to avoid reuse +} // namespace FrontierEntryState + +// Search-level outcome (design doc's Termination section), distinct from a +// single pass's per-call truncation (ReferenceChainTracker::runPass()'s +// `out_truncated`, unchanged from the original single-pass heap-walk engine): +// a pass can be truncated - budget +// or frontier cap exhausted for *that call* - without the search itself +// being ABANDONED, because there may be nothing left to do (RUNNING is still +// correct) or plenty left for the next pass to pick up. See runPass()'s own +// comment for exactly which conditions move _search_state out of RUNNING. +namespace SearchState { +constexpr u8 RUNNING = 0; // at least one more pass may still make progress +constexpr u8 COMPLETED = 1; // reachable graph fully explored within caps +constexpr u8 ABANDONED = 2; // TTL or frontier-size cap forced an incomplete stop +} // namespace SearchState + +// Records which cutoff actually moved a search from RUNNING to ABANDONED +// (runPass()'s Termination-section decision, referenceChains.cpp) - recorded +// so abandonReason() (and the T_REFERENCE_CHAIN_ABANDONED JFR event built +// from it, see buildAbandonedEvent()) can report *why*, per the design doc's +// "no silent truncation" requirement, rather than just *that* it happened. +// Values match Recording::recordReferenceChainAbandoned()'s kReasons table +// (flightRecorder.cpp) index-for-index. +namespace SearchAbandonReason { +constexpr u8 NONE = 0; // not (yet) abandoned +constexpr u8 FRONTIER_CAP = 1; // frontier-size cap hit +constexpr u8 TTL = 2; // wall-clock TTL exceeded with work still pending +// Canary candidate-discovery has made no progress for +// NO_PROGRESS_PASS_LIMIT consecutive passes. Unlike TTL above, this fires +// even while isUrgent() holds - see CANARY_NO_PROGRESS_PASS_LIMIT's own +// comment for why the ordinary TTL's !isUrgent() guard must not apply here. +constexpr u8 CANARY_STUCK = 3; +} // namespace SearchAbandonReason + +// Frontier/EdgeStore record (design doc: "Data structures" / +// "Frontier metadata storage"). Deliberately does not hold a live +// jclass/jobject: retaining either would defeat the point of using +// non-retaining JVMTI tags for frontier identity. `referrer_klass` is a +// StringDictionary id (Profiler::classMap(), profiler.h - the same +// interning table LivenessTracker uses via Profiler::lookupClass()) +// resolved from a class name string; the +// heap-walk engine populates it from GetClassSignature, and FrontierEntry +// only needs the field. +typedef struct FrontierEntry { + jlong parent_tag; // links back to the record that discovered this one + u32 referrer_klass; // StringDictionary id, 0 = unresolved/none + u32 depth; // hop count from the frontier's seed, for the hop cap + u8 state; // one of FrontierEntryState's constants + // The leak tag assigned by LivenessTracker to this specific tracked + // object, copied from the JVMTI tag at admission time. 0 = not a + // leak-tagged object (ordinary BFS admission). When non-zero, this is + // the stable correlation ID written into both ReferenceChain.targetTag + // and HeapLiveObject.leakTag — the backend joins on this field to + // match a reference chain to the specific leaking heap object it + // describes. + jlong leak_tag; + // jvmtiHeapReferenceKind of the edge that admitted this entry, but only + // meaningful when parent_tag == 0 (this entry is root-attached) - 0 (no + // JVMTI_HEAP_REFERENCE_* value is 0) for every other entry, since a + // non-root entry's own referrer edge kind is not what + // reconstructChain()'s callers want to report (they want to label the + // chain's root, not every hop). Set by heapReferenceCallback() + // (referenceChains.cpp) at insert() time. + u8 root_kind; + // Raw JVMTI class tag of THIS entry's own object, from the shared, + // process-wide allocator (classTagAllocator.h) - NOT referrer_klass above + // (a classMap dictionary id, which can differ for the same class at + // different times if that dictionary gets compacted/regenerated - see + // LivenessTracker::KlassPopulationEntry::stable_class_tag's own comment + // for the bug this was found fixing). Populated at admission time + // (admitObject()) directly from the class_tag value heapReferenceCallback()/ + // heapRootCallback() already receive as a JVMTI callback parameter - no + // extra JVMTI call needed. Stored (rather than only used transiently at + // admission time) specifically so ReferenceChainTracker:: + // seedLeakAccumulationForNewlyWatchedKlass()'s retroactive scan can read + // it back later for entries admitted long before that scan runs - a live + // JVMTI callback cannot be replayed after the fact. 0 = unresolved/none, + // same convention as referrer_klass. + jlong class_tag; + + // Retention-edge identity of the edge that admitted THIS entry, captured + // at admission time for the same cannot-replay-the-callback reason as + // class_tag above: it lets the emitted datadog.ReferenceChain name the + // field each hop is retained through, turning the bare class list into a + // readable path ("LeakHolder.SINK -> HashMap.table -> Entry.value") - + // which is the diagnostic point of the whole feature. + // + // referrer_field_index: for a FIELD/STATIC_FIELD admitting edge, the + // JVMTI-SPECIFICATION field ordinal of that field in the REFERRER's + // flattened field space - the ordinal space the heap callbacks' reference + // reference_info->field.index is defined over (JVMTI spec, + // jvmtiHeapReferenceInfoField: interface-implemented fields, then the + // superclass chain root-first, then the class's own fields, each in + // GetClassFields order - verified against HotSpot's implementation, + // jvmtiTagMap.cpp ClassFieldMap::create_map_of_*). NOT a position in any + // one GetClassFields result alone - the referrer class's FULL hierarchy is + // needed to decode it (resolveFieldEdgeName(), below). -1 = the admitting + // edge was not a field/static-field reference (or reference_info was + // unavailable). + jint referrer_field_index; + + // jvmtiHeapReferenceKind of the admitting edge for an INTERIOR hop (an + // entry with parent_tag != 0 - "this object was reached from its parent + // via this kind of edge"). Root-attached entries do not use this field + // (their edge kind IS root_kind by definition). 0 = not recorded. + u8 edge_kind; + + // Referrer's class tag when the referrer is a CLASS OBJECT rather than a + // frontier entry (a root-attached static-field admission - the referrer + // is the declaring class, parent_tag == 0 so there is no parent entry to + // read a class from). Interior hops leave 0 (their referrer class is the + // parent entry's own class_tag). Needed at emission to know WHICH + // hierarchy the referrer_field_index ordinal is defined over. + jlong referrer_class_tag; +} FrontierEntry; + +// Per-hop retention-edge identity collected by reconstructChain() alongside +// the class chain: everything needed at emission to label HOW chain[i] is +// retained (the field of chain[i+1] pointing at it, or the root edge for the +// last hop). Aligned with the leaf-to-root chain order - edges[i] describes +// the edge INTO chain[i]. +typedef struct ChainHopEdge { + // FrontierEntry::referrer_field_index/edge_kind of the entry for chain[i] + jint field_index; // -1 = not a field/static-field edge + u8 edge_kind; // admitting edge kind (root hops: root_kind) + // The referrer's raw class tag: the PARENT entry's class_tag for interior + // hops, FrontierEntry::referrer_class_tag for root-attached hops. 0 = + // unknown (label degrades to the edge kind, never a fabricated name). + jlong referrer_class_tag; +} ChainHopEdge; + +// Durability ranking for FrontierEntry::root_kind (design doc's "Fix for +// root-attribution staleness" point 1): higher +// is more durable. Used to decide whether a newly-observed root reference to +// an already-admitted, root-attached entry should replace its recorded +// root_kind rather than keeping whichever root happened to be enumerated +// first. Only the three tiers the design doc actually names are ranked with +// confidence ("static/class/CLD > JNI global > JNI local/stack local/ +// monitor"); JVMTI_HEAP_REFERENCE_THREAD and _OTHER have no documented tier +// and are conservatively bucketed with the least-durable tier rather than +// assumed durable. +inline int rootKindDurability(u8 root_kind) { + switch (root_kind) { + case JVMTI_HEAP_REFERENCE_STATIC_FIELD: + case JVMTI_HEAP_REFERENCE_SYSTEM_CLASS: + return 3; + case JVMTI_HEAP_REFERENCE_JNI_GLOBAL: + return 2; + case JVMTI_HEAP_REFERENCE_MONITOR: + case JVMTI_HEAP_REFERENCE_STACK_LOCAL: + case JVMTI_HEAP_REFERENCE_JNI_LOCAL: + case JVMTI_HEAP_REFERENCE_THREAD: + case JVMTI_HEAP_REFERENCE_OTHER: + return 1; + default: + return 0; // root_kind's own "not set"/non-root-attached value + } +} + +// True for the two root kinds the design doc calls "first observed via" +// rather than "rooted by" evidence (design doc point 2): a stack-local or +// JNI-local reference is only alive for as long as its owning frame/handle +// scope is on some thread's stack, so an entry admitted through one is +// always a candidate both for a durability upgrade (rootKindDurability() +// above) and for the softer output label (flightRecorder.cpp's +// rootKindName()) and for the bounded rotating re-expansion +// (ReferenceChainTracker::collectStaleRootKindEntriesForRotation()). +inline bool isTransientRootKind(u8 root_kind) { + return root_kind == JVMTI_HEAP_REFERENCE_STACK_LOCAL || + root_kind == JVMTI_HEAP_REFERENCE_JNI_LOCAL; +} + +// Tag-indexed slot table storing FrontierEntry metadata, modeled on +// LivenessTracker's TrackingEntry table (livenessTracker.h:21-30): CAS-safe +// doubling resize under a signal-safe SpinLock (spinLock.h), reusing its +// shared/exclusive split so reads (lookup) never race a resize. +// +// Structural difference from LivenessTracker's table: the slot index is the +// JVMTI tag value itself (tag - 1), not an externally-assigned array +// position. This works because ReferenceChainTracker::nextTag() hands out +// tags sequentially starting at 1 and never reuses one, so each +// tag maps to exactly one slot for the table's lifetime. +// +// Capacity is an explicit constructor parameter (wired from +// Arguments::_reference_chains_frontier_cap), not derived from heap +// size the way LivenessTracker sizes its table (livenessTracker.cpp:152-176) +// - the design doc explicitly flags that sizing formula as non-transferable +// to a BFS frontier (Open Question 2: frontier width is driven by per-hop +// fan-out, not an allocation sampling rate). Only the doubling-resize +// *mechanics* are reused from LivenessTracker, not its sizing heuristic. +// +// Concurrency: unlike LivenessTracker::track() (called from the allocation +// sampling hot path, which must never block), FrontierTable::insert()/ +// clear()/markEdge()/markExpanded() are only ever called from the single +// agent-owned BFS thread (design doc's Algorithm; the heap-walk engine), so +// they use the blocking exclusive lock() rather than LivenessTracker's +// non-blocking tryLockShared() bailout - exclusive, not shared, so a writer +// actually excludes a concurrent lookup() reader instead of merely +// serializing against other writers. lookup() may still be called +// concurrently from a reader walking parent_tag links (e.g. chain +// reconstruction), hence the shared lock there: shared mode only ever +// contends with other shared-mode readers, never with a writer's exclusive +// lock. +class alignas(alignof(SpinLock)) FrontierTable { +private: + // Provisional default pending empirical tuning - not benchmark-derived. + // Reuses LivenessTracker's doubling-resize *mechanics* + // (growLocked() below), but this starting size is a conservative guess, + // not scaled from LivenessTracker's own initial size (which that class + // derives from max_heap/sampling_interval, a formula the design doc + // explicitly flags as non-transferable to a BFS frontier - see + // arguments.h's DEFAULT_REFERENCE_CHAINS_FRONTIER_CAP comment). Small + // enough to avoid over-allocating for a search that never grows a wide + // frontier, large enough to avoid the first several growLocked() calls + // for an ordinary one; a future frontier-table peak-occupancy + // measurement pass is the intended way to replace this guess. + static constexpr int INITIAL_TABLE_CAPACITY = 1024; + + // mutable: capacity()/maxCapacity() below are const accessors that still + // need to take this lock to read _table_cap/_table_max_cap safely. + mutable SpinLock _table_lock; + // 1 + highest index ever inserted (informational upper bound for + // lookup(); never shrinks, since tags/slots are never reused). atomic + // (not volatile) because insert() updates it via a CAS loop concurrently + // with plain reads from size()/resetForRestart() - a volatile int mixed + // with __sync_bool_compare_and_swap has no synchronizes-with edge under + // the C++ memory model, so those plain reads and the CAS are a genuine + // data race (caught by TSAN), even though relaxed/informational + // semantics are all that's needed here. + std::atomic _table_size; + int _table_cap; + int _table_max_cap; + FrontierEntry *_table; + + // Grows _table (doubling) until it holds at least `required_cap` slots or + // _table_max_cap is reached. Must be called with _table_lock held + // exclusively. Returns false (capacity exhausted) without partially + // resizing if `required_cap` exceeds _table_max_cap. + bool growLocked(int required_cap); + +public: + // `max_cap` <= 0 disables the table (capacity() stays 0, every insert() + // reports exhaustion) - callers are expected to guard on the config flag + // before constructing one, but this makes a misconfigured cap fail safe + // rather than crash. + explicit FrontierTable(int max_cap); + ~FrontierTable(); + + FrontierTable(const FrontierTable &) = delete; + FrontierTable &operator=(const FrontierTable &) = delete; + + // Writes (parent_tag, referrer_klass, depth, state) into the slot for + // `tag` (index = tag - 1), growing the table if needed. Returns false + // without writing anything if `tag` is not positive, or the table is + // already at max_cap and still too small for this tag - the design doc's + // frontier-size-cap requirement is "stop admitting new entries and report + // it", so this reports failure to the caller rather than crashing or + // silently dropping the write. + // `root_kind` is the jvmtiHeapReferenceKind of the admitting edge - only + // meaningful when `parent_tag == 0` (see FrontierEntry::root_kind's own + // comment); callers that are not admitting a root-attached entry can + // leave it at the default 0. `class_tag` is FrontierEntry::class_tag - see + // its own comment; defaults to 0 (unresolved) so call sites that do not + // participate in leak-accumulation matching (canary pruning, test seams) + // need no change. + bool insert(jlong tag, jlong parent_tag, u32 referrer_klass, u32 depth, + u8 state = FrontierEntryState::FRONTIER, u8 root_kind = 0, + jlong class_tag = 0, + jint referrer_field_index = -1, u8 edge_kind = 0, + jlong referrer_class_tag = 0); + + // Reads the slot for `tag` into *out. Returns false (leaving *out + // untouched) if `tag` is not positive or has never been inserted. + bool lookup(jlong tag, FrontierEntry *out); + + // Runs `fn(this)` with the shared lock held for the whole call, for a + // caller that needs to look up many tags back to back (e.g. the rotation + // collectors' O(size()) sweeps in referenceChains.cpp) under ONE lock + // acquisition, instead of paying SpinLock's lock/unlock cost on every + // single lookup() call. RAII (SharedLockGuard, spinLock.h) releases the + // lock on every exit path from `fn`, including an early return - unlike a + // manual lockShared()/unlockShared() pair, a `fn` that returns early can't + // leak the lock. `fn` should only call lookupLocked() on this table, never + // another FrontierTable method that tries to take the lock again. + template void withSharedLock(Fn &&fn) const { + SharedLockGuard guard(&_table_lock); + fn(this); + } + + // Same as lookup() above, but assumes the caller already holds the shared + // lock via withSharedLock() below. + bool lookupLocked(jlong tag, FrontierEntry *out) const; + + // Marks the slot for `tag` as ABANDONED in place. This is only the + // metadata-table side of tag release (design doc's Termination section); + // the caller is still responsible for SetTag(obj, 0) via + // ReferenceChainTracker::clearTag() - clear() here does not touch JVMTI. + // No-op if `tag` was never inserted. + void clear(jlong tag); + + // Marks the slot for `tag` as EDGE in place (design doc: "on a path + // toward a target sample (EdgeStore)"). No-op if `tag` was never + // inserted. Used by reconstructChain() below to mark every hop it walks. + void markEdge(jlong tag); + + // Marks the slot for `tag` as EXPANDED in place (design doc: "expanded; + // children (if any) are in the table") - the resumed-pass counterpart to + // markEdge(): ReferenceChainTracker::expandFrontier() calls this + // once an entry's own outgoing edges have been fully visited by a + // FollowReferences(initial_object=) call, so a later + // pass's scan for pending work (which only considers FRONTIER-state + // entries) skips it. No-op if `tag` was never inserted. + void markExpanded(jlong tag); + + // Overwrites the slot for `tag`'s root_kind in place, touching no other + // field - the durability-upgrade counterpart to insert()'s one-time + // root_kind write (design doc's "opportunistic upgrade during root + // re-enumeration"). No-op if `tag` was never inserted. + // + // Callers MUST only invoke this when the update itself originates from a + // root discovery (a root callback rediscovering an already-tagged object + // as a heap root), never from an ordinary edge admission/re-expansion - + // and only on an entry that is already root-attached (parent_tag == 0). + // FrontierEntry::root_kind is documented as meaningful only when + // parent_tag == 0; this mutator does not itself touch parent_tag, so + // calling it from a non-root discovery context (e.g. an + // edge-driven re-expansion rediscovering an edge to an already-tracked, + // non-root-attached object) would silently leave a non-zero root_kind on + // an entry nothing else treats as root-attached. See + // ReferenceChainTracker::maybeUpgradeRootAttachedRootKind() (the sole + // caller) for how this is enforced. + void updateRootKind(jlong tag, u8 root_kind); + + // Set the leak tag on a frontier entry (the JVMTI tag assigned by + // LivenessTracker to this specific tracked leaking object). + void setLeakTag(jlong tag, jlong leak_tag) { + if (tag <= 0 || tag - 1 > (jlong)INT_MAX) { + return; + } + int idx = (int)(tag - 1); + _table_lock.lock(); + if (idx < _table_size) { + _table[idx].leak_tag = leak_tag; + } + _table_lock.unlock(); + } + + // Replace a shallow root-attached entry (parent_tag == 0, depth == 0) + // with a deeper chain-attached entry when the object is reached via a + // longer path. This fixes the "depth=1 chain with no holder" problem: + // an object first admitted as a JNI-local root (parent_tag == 0) gets + // its frontier entry overwritten when the static-field → ... → object + // path reaches it later with a non-zero parent_tag. + // Returns true if the entry was actually improved (new depth > old). + bool improveChain(jlong tag, jlong parent_tag, u32 referrer_klass, + u32 depth, u8 root_kind, jint referrer_field_index = -1, + u8 edge_kind = 0, jlong referrer_class_tag = 0); + + // Equal-depth re-parenting, the one case improveChain() above cannot + // express: a depth-1 entry whose current parent is a TRANSIENT root + // (stack local / JNI local - a momentarily-live frame) is re-parented to + // a DURABLE root-attached parent (static field, JNI global, thread) when + // one is seen admitting the same object at the same depth. The retention + // explanation of a depth-1 chain is entirely its root hop, so a transient + // root makes the chain noise even though the depth is legitimate for the + // real static path (the actual hotdog shape: a singleton collection is a + // depth-0 static root, its elements depth 1 - equal depth means + // improveChain() sees no improvement and the noise path would stick + // forever). Only depth-1 targets qualify: at deeper depths judging root + // durability would require walking both chains, not just two lookups. + // Returns true if the parent was swapped. + bool reparentToDurableRoot(jlong tag, jlong new_parent_tag, + u32 referrer_klass, + jint referrer_field_index = -1, u8 edge_kind = 0); + + // Walks parent_tag links starting at `target_tag` back to a root-attached + // entry (parent_tag == 0), appending each visited entry's referrer_klass + // to *out_chain in leaf-to-root order, and marking each visited entry + // EDGE via markEdge() - this table's degenerate EdgeStore (design doc: + // "a chain can be walked back from a target sample to a root by + // following parent_tag across EdgeStore records"). Returns false (leaving + // *out_chain untouched) if target_tag was never inserted. Bounds the walk + // at maxCapacity() hops as a defensive guard against a corrupted/cyclic + // parent_tag chain - nextTag() only ever hands out a strictly larger value + // than any tag already assigned (true across resumed passes too, not just + // within one), so a child's parent_tag always points at an + // already-existing, strictly smaller tag and a cycle should be + // unreachable in practice; this is not a correctness dependency. + // + // `out_root_kind` (if non-null) receives the root-attached entry's own + // FrontierEntry::root_kind - the jvmtiHeapReferenceKind of whichever edge + // first admitted this chain into the frontier, letting a caller label the + // chain with why it is reachable at all (JNI global, thread stack, static + // field, ...) instead of just how (the referrer_klass hops in *out_chain). + // + // `out_terminal` (if non-null) receives the root-attached entry itself - + // the chain's root-side end. Callers that need the ROOT TYPE as a chain + // element (buildChainEvent() appends the declaring class of a + // static-field-rooted chain) read FrontierEntry::referrer_class_tag from + // it - this table stores class tags, not StringDictionary ids, so the + // resolution stays with the tracker. + bool reconstructChain(jlong target_tag, std::vector *out_chain, + u8 *out_root_kind = nullptr, + std::vector *out_edges = nullptr, + FrontierEntry *out_terminal = nullptr); + + // Search restart (ReferenceChainTracker::restartSearch(), this class's own + // header comment): marks every slot unoccupied again without releasing + // _table's allocation - a new search's nextTag() sequence restarts at 1, + // reusing these same slot indices, so lookup()/insert() must not read back + // the previous search's now-irrelevant entries for them. Safe to call + // only once releaseSearchTags() has already cleared every live JVMTI tag + // this search owned (restartSearch()'s own caller ordering) - this method + // has no way to release tags itself, it only forgets the metadata table's + // record of them. + void resetForRestart() { + _table_lock.lock(); + _table_size.store(0, std::memory_order_relaxed); + _table_lock.unlock(); + } + + // Debug-only test seam (ReferenceChainTracker::resetSearchStateForTest()). + // Unlike resetForRestart(), which only forgets this table's occupancy, + // this discards the table's whole allocation and rebuilds it at + // `max_cap` - the only way to undo the "sized once, on the first start() + // in this JVM" capacity choice (this class's own constructor comment) + // that a differently-configured test running earlier in the same, + // no-forkEvery JVM (ProfilerTestPlugin.kt) would otherwise leave every + // later test permanently stuck with. Defined in referenceChains.cpp + // alongside the constructor it mirrors. + void resetCapacityForTest(int max_cap); + + // _table_cap/_table_max_cap are plain ints, not atomics like _table_size, + // and resetCapacityForTest() (debug-only test seam, see its own comment) + // rewrites both under _table_lock after freeing/reallocating _table. Every + // other reader of these fields (growLocked() and its callers) already + // holds _table_lock; these two accessors take it too so a concurrent + // resetCapacityForTest() during shared-JVM test overlap can't race an + // unsynchronized read here. + int capacity() const { + _table_lock.lock(); + int cap = _table_cap; + _table_lock.unlock(); + return cap; + } + int maxCapacity() const { + _table_lock.lock(); + int max_cap = _table_max_cap; + _table_lock.unlock(); + return max_cap; + } + + // Current upper bound on assigned slots (mirrors _table_size's own + // comment: "1 + highest index ever inserted"). expandFrontier() + // uses this to know how far a resumed pass's scan for FRONTIER-state + // entries needs to go. Relaxed/informational like _table_size itself: a + // concurrent insert() racing this read only makes the caller's scan + // window one tag short for this call, which self-corrects on the next + // call once _table_size has caught up. + int size() const { return _table_size.load(std::memory_order_relaxed); } +}; + +// Tag-indexed table mapping a *class* tag (see +// ReferenceChainTracker::nextClassTag() - always negative, a namespace +// disjoint from the positive FrontierTable object tags above so a raw tag +// value alone always tells the heap-walk callback which table it belongs +// to) to the StringDictionary id of that class's resolved name +// (Profiler::classMap(), the same interning table LivenessTracker uses via +// Profiler::lookupClass(), livenessTracker.cpp). +// +// Populated once per loaded class by +// ReferenceChainTracker::resolveLoadedClasses() - a GetLoadedClasses() + +// GetClassSignature() pass run *before* FollowReferences starts, specifically +// so heapReferenceCallback() (referenceChains.cpp) never needs a class-name +// lookup of its own: GetClassSignature is a JNI/Class-category call, and the +// JVMTI spec forbids Heap-callback functions like heapReferenceCallback from +// calling anything but "callback safe" functions (see the header comment +// above) - resolving names inline inside the callback is not an option. +// +// Concurrency: like FrontierTable, only ever touched by the single +// agent-owned BFS thread (design doc's Algorithm "Thread" bullet), so no locking is +// needed - unlike FrontierTable there is also no cross-thread reader to +// guard against (chain reconstruction only needs FrontierTable). +class ClassTagTable { +private: + std::unordered_map _table; + +public: + void insert(jlong class_tag, u32 dict_id) { _table[class_tag] = dict_id; } + + // Returns the StringDictionary id for `class_tag`, or 0 if it was never + // inserted (0 is StringDictionary's own "no entry" sentinel too, so this + // composes with FrontierEntry::referrer_klass's documented 0 = + // unresolved/none convention without a separate "found" out-parameter). + u32 resolve(jlong class_tag) const { + auto it = _table.find(class_tag); + return it != _table.end() ? it->second : 0; + } + + size_t size() const { return _table.size(); } -// Incremental reference-chain search driven by a dedicated JVMTI agent thread. + // Drops every cached class_tag -> dict_id mapping - used when the + // underlying StringDictionary itself was reset (see + // ReferenceChainTracker::_last_class_map_generation's comment) and every + // id here now points at a namespace that no longer exists. + void clear() { _table.clear(); } +}; + +// Singleton shape mirrors LivenessTracker (livenessTracker.h). class ReferenceChainTracker { - // Test-only accessor (referenceChains_ut.cpp), mirroring vmEntry.h's VMTestAccessor pattern: - // since instance() is a process-wide singleton, the search-lifecycle fields - // (_search_state/_search_started/etc.) would otherwise leak across separate TEST_F cases in the - // same gtest binary. + // Test-only accessor (referenceChains_ut.cpp), mirroring vmEntry.h's + // VMTestAccessor pattern: since instance() is a process-wide singleton, + // the search-lifecycle fields (_search_state/_search_started/etc.) + // would otherwise leak across separate TEST_F cases in the same gtest + // binary. The accessor resets them back to their just-constructed values + // between tests; it does not change any production behavior. friend class ReferenceChainsTestAccessor; private: bool _enabled; - // Frontier metadata table. Constructed lazily on the first start() with the flag enabled, sized - // from args._reference_chains_frontier_cap; like LivenessTracker's table (LivenessTracker's own - // table does the same) it survives stop() so it persists across multiple start/stop recording - // cycles. + // Frontier metadata table. Constructed lazily on the first + // start() with the flag enabled, sized from + // args._reference_chains_frontier_cap; like LivenessTracker's table + // (LivenessTracker's own table does the same) it survives stop() so it persists across + // multiple start/stop recording cycles. FrontierTable *_frontier; - // args._reference_chains_frontier_cap as of the most recent start() call - recorded - // unconditionally (even once _frontier already exists and start() itself skips reconstructing - // it), so resetSearchStateForTest() has something to rebuild the table at other than whatever cap - // the first start() in this JVM happened to use (see _frontier's own comment). + // args._reference_chains_frontier_cap as of the most recent start() call - + // recorded unconditionally (even once _frontier already exists and start() + // itself skips reconstructing it), so resetSearchStateForTest() has + // something to rebuild the table at other than whatever cap the first + // start() in this JVM happened to use (see _frontier's own comment). int _configured_frontier_cap; - // The args-configured per-pass budget (Arguments::_reference_chains_budget), recorded at start(). + // The args-configured per-pass budget (Arguments::_reference_chains_budget), + // recorded at start(). The urgency ramp in threadLoop() multiplies the live + // _budget while a near-OOM projection holds and restores THIS value when + // urgency clears - the live _budget cannot serve as the restore source + // because it may already carry the boost. int _configured_budget; - // Class-tag -> StringDictionary id table. Populated by resolveLoadedClasses(), read by - // heapReferenceCallback(). + // Class-tag -> StringDictionary id table. Populated by + // resolveLoadedClasses(), read by heapReferenceCallback(). Survives + // stop()/start() cycles for the same reason _frontier does - a class, + // once resolved, does not need re-resolving just because the profiler + // recording was restarted - UNLESS the underlying dictionary itself was + // reset (see _last_class_map_generation below), in which case every id + // cached here is for an id namespace that no longer exists. ClassTagTable _class_tags; - // Profiler::classMap()'s generation as of the last resolveLoadedClasses() call. + // Profiler::classMap()'s generation as of the last resolveLoadedClasses() + // call. Profiler::start() calls _class_map.clearAll() (profiler.cpp) + // whenever `reset || _start_time == 0`, which restarts that + // StringDictionary's id namespace at 1 - but a class's JVMTI-level + // class-object tag (GetTag(klass, ...)) is JVM-level state, untouched by + // that reset, so resolveLoadedClasses()'s "already tagged -> already + // resolved, skip it" check (tag == 0) would otherwise keep _class_tags + // pointing at ids from a dictionary generation that clearAll() already + // wiped. resolveLoadedClasses() compares this against + // Profiler::instance()->classMap()->generation() and, on a mismatch, + // re-resolves every loaded class's name (reusing its existing tag rather + // than assigning a new one) instead of only the untagged ones - see that + // method's own comment. Initialized to 0 (StringDictionary's own initial + // generation), not a sentinel, since a resolveLoadedClasses() call before + // any clearAll() has ever run must NOT treat that as a mismatch. u64 _last_class_map_generation; - // GetLoadedClasses() count as of the last resolveLoadedClasses() call that actually ran its - // per-class GetTag()/GetClassSignature() scan - lets that method skip the scan entirely on a - // resumed pass where the loaded-class count has not CHANGED (see resolveLoadedClasses()'s own - // comment for why this must be an equality check, not just a "grew" check: the count is not - // monotonic once class unloading is in play). + // GetLoadedClasses() count as of the last resolveLoadedClasses() call that + // actually ran its per-class GetTag()/GetClassSignature() scan - lets that + // method skip the scan entirely on a resumed pass where the loaded-class + // count has not CHANGED (see resolveLoadedClasses()'s own comment for why + // this must be an equality check, not just a "grew" check: the count is + // not monotonic once class unloading is in play). Survives stop()/start() + // cycles for the same reason _class_tags does. Written and read only from + // the single BFS thread, like _last_pass_gc_finish_epoch. Forced to -1 + // (a value class_count, always >= 0, can never equal) by a + // _last_class_map_generation mismatch, so the scan is never skipped on the + // very call that must re-resolve every already-tagged class. int _last_resolved_class_count; - // GetLoadedClasses() count as of the last runPassManualWalk() call whose admitStaticFieldRoots() - // sweep actually ran (i.e. was not skipped by the guard below) AND completed without being - // truncated. + // GetLoadedClasses() count as of the last runPassManualWalk() call whose + // admitStaticFieldRoots() sweep actually ran (i.e. was not skipped by the + // guard below) AND completed without being truncated. Distinct from + // _last_resolved_class_count even though both are populated from the same + // GetLoadedClasses() count: resolveLoadedClasses() runs once per runPass() + // unconditionally (it is cheap to skip its own per-class scan once + // unchanged), whereas admitStaticFieldRoots() re-walks EVERY loaded class + // via FollowReferences - a stop-the-world HeapWalkOperation - so + // runPassManualWalk() only calls it at all when this differs from + // resolveLoadedClasses()'s freshly-observed _last_resolved_class_count, + // i.e. only when the loaded-class set has actually changed since the last + // completed sweep. Left unset (mismatched) on a truncated sweep so the + // next pass retries rather than silently treating a still-incomplete sweep + // as done. Initialized to -1 (a value class_count, always >= 0, can never + // equal) so the very first pass always runs the sweep once. int _last_static_field_class_count; - // Index into the (per-call, app-classes-first-partitioned) loaded-class list that - // admitStaticFieldRoots() resumes from on its next call - see that method's own comment for why a - // single FollowReferences over every loaded class at once (no cursor) could never finish within - // one pass's safepoint deadline on a JVM with tens of thousands of loaded classes. + // Index into the (per-call, app-classes-first-partitioned) loaded-class + // list that admitStaticFieldRoots() resumes from on its next call - see + // that method's own comment for why a single FollowReferences over every + // loaded class at once (no cursor) could never finish within one pass's + // safepoint deadline on a JVM with tens of thousands of loaded classes. + // Wrapped back to 0 once a chunk reaches the end of the current + // GetLoadedClasses() count. Clamped to 0 if the loaded-class count shrinks + // below the cursor (classes unloaded) rather than reading out of range. int _static_field_sweep_cursor; - // Set when any chunk within the current lap (the cursor's walk from 0 back to 0) truncates. + // Set when any chunk within the current lap (the cursor's walk from 0 + // back to 0) truncates. Read when the cursor wraps: a lap that truncated + // even once must not mark _last_static_field_class_count as done - the + // next lap starts immediately (cursor is already back at 0) to keep + // retrying, same "no silent truncation" contract the untruncated case + // documents. Cleared at the start of each new lap. bool _static_field_sweep_cycle_truncated; - // Per-call cap on how many classes admitStaticFieldRoots() includes in one FollowReferences call - // - see that method's own comment. + // Per-call cap on how many classes admitStaticFieldRoots() includes in one + // FollowReferences call - see that method's own comment. Provisional and + // unbenchmarked like this subsystem's other per-pass caps (e.g. + // ROOT_KIND_ROTATION_BUDGET): small enough that building the holder array + // and walking one chunk's static fields fits comfortably inside the 5-50ms + // per-pass safepoint deadline even when a class in the chunk has an + // unusually large static-field graph, large enough that a JVM with a + // realistic loaded-class count (tens of thousands) completes a full lap in + // well under a minute of wall-clock passes. static constexpr int STATIC_FIELD_SWEEP_CHUNK_CLASSES = 512; - // Per-class cap on non-STATIC_FIELD edges admitted during one admitStaticFieldRoots() lap. + // Per-class cap on non-STATIC_FIELD edges admitted during one + // admitStaticFieldRoots() lap. STATIC_FIELD edges are always admitted + // (high-priority leak root); non-static edges (CONSTANT_POOL, INTERFACE, + // SUPERCLASS, CLASS_LOADER, ...) are admitted up to this many per class, + // then dropped for the rest of that class this lap. 32 covers a typical + // class's full constant-pool/interface set; outlier classes are bounded + // so they cannot blow the chunk's safepoint deadline. See + // PassContext::_class_other_cap's own comment for the admission logic. static constexpr int STATIC_FIELD_SWEEP_NON_STATIC_CAP_PER_CLASS = 32; - // GC epoch counters, bumped from the JVMTI GC callbacks and read by shouldRunPass(). + // "GC just happened" signals. Bumped only from onGCStart()/onGCFinish(); + // gcFinishEpoch() is now read by shouldRunPass() as one of the + // two pass-scheduling triggers (design doc's Triggering section). volatile u64 _gc_start_epoch; volatile u64 _gc_finish_epoch; - // Monotonically increasing tag source for frontier objects. + // Monotonically increasing tag source for frontier objects. 0 is reserved + // (JVMTI convention: an untagged object reads back tag 0, and + // SetTag(obj, 0) clears a tag), so this starts at 1. Always hands out + // positive values - see nextClassTag() below for why classes use a + // disjoint (negative) range instead of sharing this counter. volatile jlong _next_tag; - // Per-pass tunables, copied from Arguments in start(). + // Per-pass tunables, copied from Arguments in start() (design doc: Open + // Question 2 defaults, from the config-flag scaffolding). A future + // measurement pass will decide whether/how these can change between passes + // of the same search; for now this only needs one fixed value per + // start()/stop() cycle, exactly like LivenessTracker's _subsample + // (livenessTracker.h). int _hop_cap; int _budget; - // Larger one-shot edge budget for the search's root-seeded first pass. + // Edge budget for just the search's one-shot, root-seeded first pass + // (runPass()'s !_search_started branch) - copied from + // Arguments::_reference_chains_first_pass_budget in start(), auto-scaled + // from _budget (AUTO_FIRST_PASS_BUDGET_MULTIPLIER, capped at + // AUTO_FIRST_PASS_BUDGET_CAP) when unset (0), rather than falling back to + // plain _budget: a steady-state per-pass budget sized for cheap incremental + // expansion truncates a cold root-seeded walk of a real JVM's object graph + // long before it reaches anything interesting - the exact trap a + // steady-state budget causes on a cold, large object graph. Only the + // first pass's own edge budget is this + // large - runPassManualWalk()'s IterateOverReachableObjects root/stack-ref + // enumeration itself reruns on every pass, first or resumed (see its own + // comment); already-admitted roots short-circuit cheaply via + // admitObject()'s ALREADY_ADMITTED check, so a root this pass doesn't + // reach before the budget runs out is still picked up by a later pass, not + // permanently lost. Unlike _budget/_effective_budget, this is spent at + // most once per search, not once per pass, so a much larger ceiling is + // affordable without the per-pass pacing controller (updatePacing()) ever + // seeing it - runPass() deliberately excludes the first pass's own + // duration from that signal (see runPass()'s own comment) so a large + // first-pass cost cannot throttle down every cheap expansion pass that + // follows. int _first_pass_budget; - // Wall-clock TTL for one search; <= 0 disables the cutoff. + // Wall-clock TTL, copied from Arguments in start() (design doc's + // Termination section: "a hard cap on passes-per-search or wall-clock TTL + // from first observation"). This implements the TTL half of that + // "or" - the config-flag scaffolding only added a TTL sub-option (no + // separate pass-count cap), and Open Question 2 leaves the choice between + // the two open pending a future measurement pass. <= 0 disables the TTL + // cutoff (a search can only still end via the frontier-size cap or natural + // completion). long _ttl_ms; - // Pause-time pacing controller: pause-time-SLO ceiling copied from Arguments in start() - // (Arguments::_reference_chains_pause_target_ms) - the single pause-time-SLO target, in place of - // guessing _budget/PASS_CADENCE_NS directly. + // Pause-time pacing controller: pause-time-SLO ceiling copied from + // Arguments in start() (Arguments::_reference_chains_pause_target_ms) - + // the single pause-time-SLO target, in place of guessing + // _budget/PASS_CADENCE_NS directly. Used only to (re)construct _pause_pid + // in start(); updatePacing() itself never reads it again, since it lives + // inside _pause_pid's own _target once constructed. long _pause_target_ms; - // Runtime-adjusted pause target: when isUrgent(), this is bumped to URGENT_PAUSE_TARGET_MS so - // each pass can explore more edges. + // Runtime-adjusted pause target: when isUrgent(), this is bumped + // to URGENT_PAUSE_TARGET_MS so each pass can explore more edges. + // Restored to _pause_target_ms (the configured default) once + // urgency clears. The PID controller (_pause_pid) is + // reconstructed whenever this changes so its ceiling tracks + // the new target. long _effective_pause_target_ms; - // Passes since the frontier last grew. Reset to 0 whenever _frontier->size() increases (new - // entries admitted). + // Passes since the frontier last grew. Reset to 0 whenever + // _frontier->size() increases (new entries admitted). + // When this exceeds NO_PROGRESS_PASS_LIMIT without + // the frontier growing, the search is genuinely + // stuck (not just slow) and is abandoned. int _passes_since_last_progress; - // Passes since _candidate_found_bits last changed (a candidate was newly found, or a new - // candidate was admitted into a slot). + // Passes since _candidate_found_bits last changed (a candidate was newly + // found, or a new candidate was admitted into a slot). Distinct from + // _passes_since_last_progress above: that one tracks whole-graph frontier + // growth, which keeps resetting to 0 for as long as there is any unvisited + // reachable object left, regardless of whether the canary's specific + // candidates are ever pruned - so it never reflects "the canary search + // itself is stuck", only "the graph walk is stuck". Read by runPass()'s + // canary-stuck check, which (unlike the ordinary TTL check just above it) + // is deliberately NOT suppressed by isUrgent(): the isUrgent() TTL + // suppression exists so a search that's still making real progress isn't + // killed just because the process is close to OOM, but a canary search + // that has made zero candidate-discovery progress for + // CANARY_NO_PROGRESS_PASS_LIMIT consecutive passes is provably not + // converging - continuing to run it at urgency-boosted budget/cadence only + // burns STW pause budget the rest of the process needs during the same + // OOM approach this search was launched to diagnose. int _passes_since_last_candidate_progress; - // candidate_count + popcount(found_bits) as of the last pass this was updated. + // candidate_count + popcount(found_bits) as of the last pass this was + // updated. A monotonic non-decreasing marker: _candidate_count only grows + // (pollWatchedTargets()'s admission loop never retires a slot) and + // found_bits only gains bits (heapReferenceCallback() only ever ORs a bit + // in) for as long as a search is RUNNING, so a rising sum means real + // canary progress happened since the last check; an unchanged sum means + // none did. int _last_candidate_progress_mark; - // Canary-lane pass pacing, work-scaled: the chase's inter-pass spacing is _canary_backoff_mult x - // _canary_pass_ema_ms - a multiple of what a pass actually COSTS, not a fixed wall-clock - // constant. + // Canary-lane pass pacing, work-scaled: the chase's inter-pass spacing is + // _canary_backoff_mult x _canary_pass_ema_ns - a multiple of what a pass + // actually COSTS, not a fixed wall-clock constant. The multiplier starts + // at 1 (back-to-back: the next pass starts as soon as the last one ended, + // which a cheap pass makes harmless by construction), doubles on each + // pass with NO candidate progress up to CANARY_BACKOFF_MULT_MAX, and + // resets to 1 on any candidate progress (a candidate found or a new + // candidate admitted). shouldRunPass() holds a canary pass off while + // now - _last_canary_pass_ns is below the spacing. + // + // Why work-scaled rather than a fixed cap: a fixed cap only binds when + // it exceeds the pass's own duration - measured live on hotdog (rounds + // 3-4), an un-findable candidate held the chase open for 32 minutes at + // ~88 passes/min (a full core; round 4's passes ran 0.7-4s each, so a 1s + // cap would have changed nothing at all - the loop is work-bound, never + // sleep-bound, when the pass itself exceeds the cap). Scaled against the + // pass's measured cost, the burn bound is structural: at the multiplier + // cap the chase spends <= 1/CANARY_BACKOFF_MULT_MAX of a core on pass + // work, whatever that work is - ~6% on the pod, and proportionally less + // blocking for a deep-but-cheap chase whose passes cost milliseconds + // (those keep a dense, fast-resolving chase at the same multiplier). int _canary_backoff_mult; - // 0.8/0.2 EMA of each pass's whole-call wall duration in ms - // (TSC::ticks_to_millis(pass_wall_ticks), runPass()), updated every pass - kept warm regardless - // of canary state so a chase that opens on a known-cost crawl sizes correctly from its first - // held-off decision. - u64 _canary_pass_ema_ms; - // End-of-pass timestamp of the last pass that ran with a canary chase still open (the reference - // point for the spacing above). + // 0.8/0.2 EMA of each pass's whole-call wall duration in NANOSECONDS + // (runPass(), from pass_wall_ticks via the TSC frequency), updated every + // pass - kept warm regardless of canary state so a chase that opens on a + // known-cost crawl sizes correctly from its first held-off decision. + // Nanoseconds, not milliseconds: an integer-ms EMA truncates every + // sub-5ms pass's 0.2-weighted contribution to zero, pinning the EMA at 0 + // for a chase whose passes are all cheap - spacing 0 means the backoff + // gate never holds anything off, silently removing the burn bound + // exactly for the fast-pass chases it exists to pace. + u64 _canary_pass_ema_ns; + // End-of-pass timestamp of the last pass that ran with a canary chase + // still open (the reference point for the spacing above). 0 = no canary + // pass has run since the search (re)started. u64 _last_canary_pass_ns; - // Whether threadLoop()'s OOM urgency ramp is currently active, set each loop iteration BEFORE - // shouldRunPass() (same thread, no atomics needed). + // Whether threadLoop()'s OOM urgency ramp is currently active, set each + // loop iteration BEFORE shouldRunPass() (same thread, no atomics + // needed). While set, shouldRunPass()'s canary-backoff gate is bypassed: + // imminent OOM is the one regime where the chase is deliberately allowed + // to burn budget back-to-back, exactly as before the backoff existed. bool _oom_ramp_active; // How many consecutive times in a row runPass()'s canary-stuck check - // (CANARY_NO_PROGRESS_PASS_LIMIT below) has abandoned this same candidate-chase sequence. + // (CANARY_NO_PROGRESS_PASS_LIMIT below) has abandoned this same + // candidate-chase sequence. Never reset by restartSearch() - it must + // survive across restarts to make the escalation in + // canaryStuckPassLimit() actually widen with repeated failures. Reset to + // 0 whenever a search leaves RUNNING for a reason other than + // CANARY_STUCK (natural completion, candidate-complete, frontier cap, + // TTL) - see the terminal-state block in runPass(). int _canary_stuck_restart_count; - // Canary-search candidate set: LivenessTracker-flagged leak klasses this - // tracker chases, one slot per klass. The retired marker-tag pre-tagging - // (each candidate's representative pre-tagged MARKER_TAG_BASE - i for an - // identity match in the walk) was replaced by leak-tag interception - - // LivenessTracker tags specific tracked instances, and discovery records - // _candidate_found_bits/_candidate_frontier_tags when one is resolved. + // Canary-search candidate set: pre-tagged with distinct + // marker tags (MARKER_TAG_BASE - i) before the walk, applied to each + // candidate's specific representative object (identity match) - matching + // by class alone would let the walk record a chain for an unrelated, + // possibly short-lived, instance of the same class instead of the one + // LivenessTracker actually flagged as growing. + // _candidate_found_bits is a packed bitmap (bit i = candidate i found). + // _candidate_tags[i] holds the marker tag (MARKER_TAG_BASE - i) for candidate i. + // _candidate_frontier_tags[i] holds the frontier tag assigned by + // heapReferenceCallback() when it pruned the candidate (equal to the + // marker tag itself, since the marker is already a unique table key). + // Reset by resetSearchStateForTest(). + // + // MAX_LEAK_CANDIDATES_FROM_LT must match + // LivenessTracker::MAX_LEAK_CANDIDATES. + // Duplicated here to avoid a heavy include chain. static constexpr int MAX_LEAK_CANDIDATES_FROM_LT = 5; // How many klass_ids _watched_leak_klass_ids tracks at once - matches // LivenessTracker::MAX_LEAK_CANDIDATES (livenessTracker.h), the cap - // LivenessTracker::topKlassesByGenerationCount() itself already enforces; duplicated here for the - // same reason MAX_LEAK_CANDIDATES_FROM_LT above already duplicates it, rather than depending on a - // private LivenessTracker constant. + // LivenessTracker::topKlassesByGenerationCount() itself already enforces; + // duplicated here for the same reason MAX_LEAK_CANDIDATES_FROM_LT above + // already duplicates it, rather than depending on a private LivenessTracker + // constant. static constexpr int MAX_WATCHED_LEAK_KLASSES = 5; - int _candidate_count; - u64 _candidate_found_bits; - // klass_id occupying each slot, so pollWatchedTargets() can tell whether a klass_id - // selectLeakCandidates() returns this poll already has a slot (and must not be - // re-tagged/re-admitted) or is new (and should be admitted into the next free slot). - u32 _candidate_klass_ids[MAX_LEAK_CANDIDATES_FROM_LT]; - jlong _candidate_frontier_tags[MAX_LEAK_CANDIDATES_FROM_LT]; - // Per-candidate chain link recorded at pruning time: parent_tag (referrer's frontier tag, - // positive) and referrer_klass. - jlong _candidate_parent_tags[MAX_LEAK_CANDIDATES_FROM_LT]; - u32 _candidate_referrer_klasses[MAX_LEAK_CANDIDATES_FROM_LT]; - u32 _candidate_depths[MAX_LEAK_CANDIDATES_FROM_LT]; + // Zero-initialized in-class: the constructor's initializer list covered + // every scalar lifecycle field EXCEPT this candidate state, leaving it + // uninitialized until the first pollWatchedTargets()/ + // resetSearchStateForTest() ran - and the BFS thread reads some of it + // (e.g. _candidate_count in shouldRunPass()'s stuck-candidate checks) + // before any poll under unusual startup orders. In-class init makes the + // zero state unconditional. + int _candidate_count = 0; + u64 _candidate_found_bits = 0; + // klass_id occupying each slot, so pollWatchedTargets() can tell whether a + // klass_id selectLeakCandidates() returns this poll already has a slot + // (and must not be re-tagged/re-admitted) or is new (and should be + // admitted into the next free slot). Slots are never retired or reused + // once assigned for the lifetime of a search - see pollWatchedTargets()'s + // admission loop for why. + u32 _candidate_klass_ids[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + jlong _candidate_tags[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + jlong _candidate_frontier_tags[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + // Per-candidate chain link recorded at pruning time: + // parent_tag (referrer's frontier tag, positive) and referrer_klass. + // Used by buildCanaryChainEvent() to reconstruct the + // chain without a frontier table lookup on the + // negative marker tag (which lookup() rejects). + jlong _candidate_parent_tags[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + u32 _candidate_referrer_klasses[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + u32 _candidate_depths[MAX_LEAK_CANDIDATES_FROM_LT] = {}; // Per-SLOT snapshot of the qualifying allocating-thread tids - // LivenessTracker::selectLeakCandidates() reported for that slot's klass this poll, refreshed by - // pollWatchedTargets() (zeroed first, then filled from the poll's candidates so a klass that - // stops qualifying stops walking its tids too). + // LivenessTracker::selectLeakCandidates() reported for that slot's klass + // this poll, refreshed by pollWatchedTargets() (zeroed first, then filled + // from the poll's candidates so a klass that stops qualifying stops + // walking its tids too). The BFS-thread walk phases run between polls + // under the same engine serialization and read this snapshot to reach + // the exact (klass, tid) scope tagLeakInstances() tags - see + // walkCandidateThreadLocals()'s own comment for why that reach matters. + // Sized generously above LivenessTracker's + // KlassCandidate::MAX_QUALIFYING_TIDS (= 8, itself + // KlassPopulationEntry::MAX_TID_TRENDS): the snapshot loop clamps, so + // this bound only needs to be "large enough" (same pattern as + // kMaxWatchedCandidates in pollWatchedTargets()). static constexpr int MAX_CANDIDATE_QUALIFYING_TIDS = 16; jint _candidate_qualifying_tids[MAX_LEAK_CANDIDATES_FROM_LT] - [MAX_CANDIDATE_QUALIFYING_TIDS]; - int _candidate_qualifying_tid_count[MAX_LEAK_CANDIDATES_FROM_LT]; - - // tid -> JNI global reference to the live java.lang.Thread object, fed from - // Profiler::onThreadStart/onThreadEnd (registerThreadObject()/unregisterThreadObject()). + [MAX_CANDIDATE_QUALIFYING_TIDS] = {}; + int _candidate_qualifying_tid_count[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + + // tid -> JNI global reference to the live java.lang.Thread object, fed + // from Profiler::onThreadStart/onThreadEnd + // (registerThreadObject()/unregisterThreadObject()). A strong global + // reference to a live thread's Thread object does not distort reachability + // (a running thread's Thread object is reachable via the VM anyway) and is + // released at ThreadEnd, so it cannot outlive the thread. Walked by + // walkCandidateThreadLocals() as the FollowReferences initial object of a + // bounded descend walk - the candidate-scoped reach mechanism (see + // descendFromAnchor()'s own comment). Mutex-guarded rather than relying + // on the engine lock: ThreadStart/End fire on arbitrary JVM threads, + // outside the engine serialization the walk phases run under. Mutex _thread_objects_lock; std::unordered_map _thread_objects; - // Global refs of ended threads awaiting deletion. unregisterThreadObject() must NOT - // DeleteGlobalRef() directly: walkCandidateThreadLocals() copies the jobject out of - // _thread_objects under _thread_objects_lock, releases the lock, and can still be using it as a - // FollowReferences anchor when a concurrent ThreadEnd erases the entry - deleting there would be - // JNI use-after-free (once deleted, a global ref is invalid for every other JNI call). + // Global refs of ended threads awaiting deletion. unregisterThreadObject() + // must NOT DeleteGlobalRef() directly: walkCandidateThreadLocals() copies + // the jobject out of _thread_objects under _thread_objects_lock, releases + // the lock, and can still be using it as a FollowReferences anchor when a + // concurrent ThreadEnd erases the entry - deleting there would be JNI + // use-after-free (once deleted, a global ref is invalid for every other + // JNI call). The erasing thread only enqueues; releaseEndedThreadRefs() + // drains the list on the BFS thread, at points where no walk phase holds + // a copied ref. + // Locking: guarded by _thread_objects_lock, the SAME mutex as the map + // above (the enqueue happens in unregisterThreadObject()'s erased-entry + // path, already holding it; the drain takes it to swap the vector out). + // This was previously undocumented - the field sat next to an explicitly + // mutex-guarded member with no stated discipline of its own. std::vector _thread_refs_pending_delete; - // Auto-marked instances: when the BFS walk discovers ANY object whose class matches a watched - // leak class (not just the pre-tagged representative), its frontier tag is recorded here so - // pollWatchedTargets() can build chain events for all of them. + // Auto-marked instances: when the BFS walk discovers ANY object whose + // class matches a watched leak class (not just the pre-tagged + // representative), its frontier tag is recorded here so pollWatchedTargets() + // can build chain events for all of them. A leaking class typically has + // many live instances, and each one's reference chain is independently + // useful for diagnosis — the pre-tagged representative is just one sample, + // and its chain may differ from other instances' chains (different parents, + // different retention paths). Fixed-size per slot to avoid heap allocation + // in the callback (safepoint context). When the per-slot array fills, + // further instances are silently dropped (the representative + up to + // MAX_DISCOVERED_INSTANCES_PER_CLASS others is still far more coverage + // than the single-representative design it replaces). static constexpr int MAX_DISCOVERED_INSTANCES_PER_CLASS = 8; jlong _candidate_discovered_tags[MAX_LEAK_CANDIDATES_FROM_LT] - [MAX_DISCOVERED_INSTANCES_PER_CLASS]; - int _candidate_discovered_count[MAX_LEAK_CANDIDATES_FROM_LT]; - - // klass_ids from LivenessTracker::topKlassesByGenerationCount() (a faster, un-hysteresis-gated - // ranking than the canary candidate set above - see that method's own comment), refreshed once - // per BFS-thread tick but only once hasLeakSignal() has already fired via the slower, - // hysteresis-gated selectLeakCandidates() path (per design discussion: this whole mechanism only - // cranks once the trend detector has already triggered, so it never needs to wait out that same - // hysteresis a second time on its own). + [MAX_DISCOVERED_INSTANCES_PER_CLASS] = {}; + int _candidate_discovered_count[MAX_LEAK_CANDIDATES_FROM_LT] = {}; + + // klass_ids from LivenessTracker::topKlassesByGenerationCount() (a faster, + // un-hysteresis-gated ranking than the canary candidate set above - see + // that method's own comment), refreshed once per BFS-thread tick but only + // once hasLeakSignal() has already fired via the slower, hysteresis-gated + // selectLeakCandidates() path (per design discussion: this whole mechanism + // only cranks once the trend detector has already triggered, so it never + // needs to wait out that same hysteresis a second time on its own). + // Consulted at admission time (heapReferenceCallback()) to decide whether + // a newly-admitted object's class is worth tracking for rotation priority + // - see _leak_signature_totals/_leak_parent_fanout's own comments for what + // that tracking actually does. Whenever a klass_id newly enters this + // array (was not present in the previous refresh), pollWatchedTargets() + // also calls seedLeakAccumulationForNewlyWatchedKlass() for it once - see + // that method's own comment for why admission-time tracking alone cannot + // see objects admitted before watching started. _watched_leak_klass_count + // entries are valid; the rest of the array is unspecified. u32 _watched_leak_klass_ids[MAX_WATCHED_LEAK_KLASSES]; int _watched_leak_klass_count = 0; // Packs a (leaf_klass_id, parent_class_id) pair into one map key for - // _leak_signature_totals/_leak_signature_prev_totals below - both are StringDictionary ids (u32), - // so this never loses information and avoids defining a custom hash/equality functor for a - // 2-field struct key. + // _leak_signature_totals/_leak_signature_prev_totals below - both are + // StringDictionary ids (u32), so this never loses information and avoids + // defining a custom hash/equality functor for a 2-field struct key. static u64 leakSignatureKey(u32 leaf_klass_id, u32 parent_class_id) { return ((u64)leaf_klass_id << 32) | (u64)parent_class_id; } // Tier 1 of the leak-accumulation rotation design (see - // collectLeakAccumulationCandidatesForRotation()'s own comment for the full design): aggregate, - // per (leaf_klass_id, parent_class_id) signature - not per object - how many admitted children of - // that leaf klass_id have been observed under a parent of that class. + // collectLeakAccumulationCandidatesForRotation()'s own comment for the + // full design): aggregate, per (leaf_klass_id, parent_class_id) signature + // - not per object - how many admitted children of that leaf klass_id + // have been observed under a parent of that class. Incremented at + // admission time (heapReferenceCallback(), O(1) per matching new + // admission - see _watched_leak_klass_ids' own comment), so this reflects + // the CURRENT cumulative total, never decreasing within a search. Small: + // bounded by (distinct leaf klass_ids ever watched) x (distinct parent + // classes ever seen holding one of them) - nowhere near the whole + // table's size, even for a common leaf class held by many unrelated + // parents, because it aggregates BY CLASS, not by individual parent + // object (that finer granularity is _leak_parent_fanout below). std::unordered_map _leak_signature_totals; - // Snapshot of _leak_signature_totals as of the END of the previous pass - runPassManualWalk() - // computes each signature's delta (totals - this) to rank signatures by growth before rolling - // this forward to the current totals for the next pass's comparison. + // Snapshot of _leak_signature_totals as of the END of the previous pass - + // runPassManualWalk() computes each signature's delta (totals - this) to + // rank signatures by growth before rolling this forward to the current + // totals for the next pass's comparison. A signature with no prior + // snapshot (brand new this pass) is treated as prev_total == 0, so its + // delta is its whole total - correctly ranks a suddenly-appearing + // signature as growing, without a special first-seen case. std::unordered_map _leak_signature_prev_totals; - // Tier 2 of the leak-accumulation rotation design: per PARENT TAG (not per class), how many - // admitted children of a watched leaf klass_id this specific parent object holds, plus which - // signature it belongs to (so rotation-selection can filter to the pass's winning signature - // without a second lookup). + // Tier 2 of the leak-accumulation rotation design: per PARENT TAG (not + // per class), how many admitted children of a watched leaf klass_id this + // specific parent object holds, plus which signature it belongs to (so + // rotation-selection can filter to the pass's winning signature without a + // second lookup). Incremented at the same admission-time hook as + // _leak_signature_totals above. Bounded by however many distinct parent + // objects have ever been observed holding a watched leaf klass_id - + // small in practice even for a common leaf type, since it is scoped to at + // most MAX_WATCHED_LEAK_KLASSES specific klass_ids, not every collection- + // shaped class in the JVM (the structural heuristic this design replaced - + // see git history for why that was measured and found not to work). struct LeakParentFanoutEntry { u64 signature_key; u32 fanout; @@ -238,49 +1131,130 @@ class ReferenceChainTracker { std::unordered_map _leak_parent_fanout; // Rotating skip-count over _leak_parent_fanout's iteration order for - // collectStaleExpandedEntriesForRotation()'s leak-parent-priority tier - advances by the number - // of parents selected each pass so, across passes, every fanout parent gets re-walked within - // ceil(fanout_size/budget) passes instead of only whichever entries the hash iteration happens to - // yield first. + // collectStaleExpandedEntriesForRotation()'s leak-parent-priority tier - + // advances by the number of parents selected each pass so, across passes, + // every fanout parent gets re-walked within ceil(fanout_size/budget) + // passes instead of only whichever entries the hash iteration happens to + // yield first. Same single-BFS-thread access as the fanout map itself. u64 _leak_parent_rotation_cursor = 0; - // Pause-time pacing controller: the actual per-pass budget runPass() passes to - // FollowReferences/expandFrontier(), replacing _budget's old role as a literal per-pass value - - // _budget above becomes this controller's ceiling instead (never exceeded, see updatePacing()), - // while this field is what updatePacing() actually raises/lowers pass to pass. + // Pause-time pacing controller: the actual per-pass budget runPass() passes + // to FollowReferences/expandFrontier(), replacing _budget's old role as a + // literal per-pass value - _budget above becomes this controller's ceiling + // instead (never exceeded, see updatePacing()), while this field is what + // updatePacing() actually raises/lowers pass to pass. Starts at _budget in + // start(), so a tracker that has not measured a pass yet behaves exactly as + // before the pacing controller was added. int _effective_budget; - // Pause-time pacing controller: the actual fallback cadence shouldRunPass()/threadLoop() compare - // against, replacing the fixed PASS_CADENCE_NS constant below in that role once updatePacing() - // starts adjusting it - see PASS_CADENCE_NS's own comment for why that constant survives as this - // field's starting value rather than being deleted outright. + // Pause-time pacing controller: the actual fallback cadence + // shouldRunPass()/threadLoop() compare against, replacing the fixed + // PASS_CADENCE_NS constant below in that role once updatePacing() starts + // adjusting it - see PASS_CADENCE_NS's own comment for why that constant + // survives as this field's starting value rather than being deleted + // outright. u64 _effective_cadence_ns; - // Budget-borrowing: extra headroom updatePacing() has temporarily granted above _budget's own - // ceiling, earned by a sustained run of comfortably- under-target passes (see - // BORROW_WARMUP_PASSES's own comment). + // Budget-borrowing: extra headroom updatePacing() has temporarily granted + // above _budget's own ceiling, earned by a sustained run of comfortably- + // under-target passes (see BORROW_WARMUP_PASSES's own comment). This is + // the one exception to "_budget is never exceeded" (_effective_budget's + // own comment) - it exists because a fast-growing frontier (a real + // leaking-cache workload, not just a synthetic one) can otherwise starve + // under a steady-state budget sized for ordinary incremental expansion, + // never converging within this search's TTL even though pause time has + // visible headroom to spare. Revoked immediately (reset to 0, see + // updatePacing()) the moment a pass is no longer comfortably under target, + // so a search that starts abusing its pause-time budget loses the + // borrowed headroom before the very next pass - _budget itself remains the + // hard ceiling in that case, same as before this field existed. int64_t _borrowed_budget; - // Budget-borrowing: number of consecutive passes (since the last reset) that came in comfortably - // under _pause_target_ms (see BORROW_UNDER_TARGET_FRACTION). + // Budget-borrowing: number of consecutive passes (since the last reset) + // that came in comfortably under _pause_target_ms (see + // BORROW_UNDER_TARGET_FRACTION). Reset to 0 the moment a pass does not + // qualify - see _borrowed_budget's own comment on why this must be a + // consecutive-run counter, not a cumulative one: a single expensive pass + // means the frontier is not, in fact, converging with room to spare, and + // borrowing more budget for the next pass on the strength of an unrelated + // earlier streak would defeat the point of gating growth on *sustained* + // headroom at all. int _consecutive_under_target_passes; - // Pause-time pacing controller: this tracker's own PidController instance - see updatePacing() - // below for the full mechanism, and PASS_CADENCE_NS's neighboring constants for why its gains are - // not copied from ObjectSampler/ MallocTracer/NativeSocketSampler's shared triple. + // Pause-time pacing controller: this tracker's own PidController instance - + // see updatePacing() + // below for the full mechanism, and PASS_CADENCE_NS's neighboring + // constants for why its gains are not copied from ObjectSampler/ + // MallocTracer/NativeSocketSampler's shared triple. Placeholder-constructed + // here (target=1, unit gains); start() reconstructs it once + // _pause_target_ms is known, mirroring RateLimiter's own + // placeholder-then-reconstruct pattern (rateLimiter.h's + // `_pid{1, 1.0, 1.0, 1.0, 1, 1.0}` member default, replaced in + // RateLimiter::start()). PidController _pause_pid; - // Self-calibrating adaptive batch sizing for GetObjectsWithTags (see expandFrontier()'s own - // comment). + // Self-calibrating adaptive batch sizing for GetObjectsWithTags (see + // expandFrontier()'s own comment). GetObjectsWithTags iterates the whole + // JVMTI tag map per call, so its cost has a batch-independent floor that + // grows with the frontier (~20-25ms at a 225-245k-entry map, measured live + // on hotdog) plus a small per-searched-tag component (measured live: + // batch 8 -> 27.4ms, batch 72 -> 36.1ms, i.e. ~0.12ms per extra tag on a + // ~25ms floor). Two earlier designs both collapsed: + // - per-TAG EMA (ema = elapsed / batch_size): the floor dominates at + // small batch sizes, so a smaller batch INFLATES the per-tag cost, + // shrinking the batch further (observed live: ~400 -> 2). + // - per-CALL AIMD against a FIXED budget: once the map grows enough that + // the floor alone exceeds the budget (25ms budget vs ~27ms floor at a + // 243k-entry map), every call "overran" regardless of batch size, so + // AIMD ratcheted to GOTW_MIN_BATCH and stayed there - measured live + // batch=8 on every call while batch=72 cost only +36% for 9x the + // objects. + // In the floor-dominated regime the right move is the OPPOSITE of + // shrinking: a bigger batch amortizes the floor. The control law is a + // direct proportion - scale the batch so ONE call fills the remaining + // wall-clock window: + // ema_call_ns = ema × 0.8 + measured × 0.2 (per-CALL, not per-tag) + // window_ns = remaining _pass_deadline_ns (nominal GOTW_CPU_BUDGET_NS + // when no deadline is set) + // batch = clamp(batch × window_ns / ema_call_ns, MIN, MAX) + // Converges upward while calls come in under the window (a near-free + // call scales the batch to GOTW_MAX_BATCH), shrinks as the window + // drains so tail calls still fit, and tracks the floor automatically + // as the tag map grows or shrinks. + // + // The window itself is computed by gotwWindowNs() below, which widens it + // under backlog pressure when the measured per-call floor already + // exceeds the remaining pass window - see that method's own comment. size_t _gotw_batch_size = 0; // 0 = unset, use GOTW_INITIAL_BATCH_SIZE u64 _gotw_ema_call_ns = 0; // EMA of per-call elapsed, 0 = unset - // Nominal per-call window for the proportional batch control above when no phase deadline is set - // (expandFrontier()'s window is the REMAINING deadline, which the phases refresh per invocation). + // Nominal per-call window for the proportional batch control above when + // no phase deadline is set (expandFrontier()'s window is the REMAINING + // deadline, which the phases refresh per invocation). Also doubles as a + // CPU-overhead sanity target: ~25ms per call at ~2-4 calls/pass is + // ~50-100ms/sec of CPU overhead on one core, acceptable for a background + // search thread on a multi-core machine. static constexpr u64 GOTW_CPU_BUDGET_NS = 25000000; // 25ms - // Effective window for the proportional batch control (expandFrontier() calls this with the - // remaining pass deadline and the depth of the lane the next GetObjectsWithTags call will drain). + // Effective window for the proportional batch control (expandFrontier() + // calls this with the remaining pass deadline and the depth of the lane + // the next GetObjectsWithTags call will drain). The remaining-deadline + // window can be SMALLER than the measured per-call floor - measured live + // on hotdog (round 4, ev-leaktag-onpod-round4): a ~10ms expand window vs + // a 14-41ms floor at a 242k-entry tag map collapsed the proportional law + // to GOTW_MIN_BATCH on every call (batch=8 forever) while the lane it + // was draining held 127k entries - a self-inflicted ~120-200 + // objects/min drain. In that regime the floor is paid by EVERY call + // regardless of batch size, so the right move is to AMORTIZE it: when + // the lane is deeper than GOTW_BACKLOG_MIN_DEPTH and the floor exceeds + // the remaining window, widen the window to EMA x + // GOTW_BACKLOG_WINDOW_MULT so the proportional law sizes the batch UP + // (batch = calib x window/ema = calib x MULT - the floor's per-tag + // surcharge, ~0.12ms/tag measured, stays the real limit). The widening + // only sizes the NEXT batch; _pass_deadline_ns itself is unchanged, so + // the per-pass overrun is bounded by one widened call, and it never + // applies to shallow lanes (rotation fast-lane batches stay + // deadline-sized, keeping targeted re-walks cheap and frequent). u64 gotwWindowNs(u64 remaining_ns, size_t lane_depth) const { u64 window_ns = remaining_ns != 0 ? remaining_ns : GOTW_CPU_BUDGET_NS; if (lane_depth >= GOTW_BACKLOG_MIN_DEPTH && @@ -292,81 +1266,210 @@ class ReferenceChainTracker { } // Lane depth beyond which gotwWindowNs()'s backlog widening applies. + // Above this, the lane itself proves that draining it matters more than + // keeping each pass strictly inside its remaining window; below it the + // ordinary proportional law applies unchanged. Chosen an order of + // magnitude above the rotation inflow per pass (~25-64 entries) so + // ordinary rotation cycling never widens the window. static constexpr size_t GOTW_BACKLOG_MIN_DEPTH = 4096; - // How many measured per-call floors one widened window may cost - see gotwWindowNs() above. + // How many measured per-call floors one widened window may cost - see + // gotwWindowNs() above. 3x converges the batch upward by 3x per call in + // the floor regime (8 -> 24 -> 72 -> 216 -> GOTW_MAX_BATCH) while keeping + // a single call's overrun bounded to a small multiple of what the floor + // already forced. static constexpr u64 GOTW_BACKLOG_WINDOW_MULT = 3; - // Conservative initial batch_size before the first GetObjectsWithTags measurement. + // Conservative initial batch_size before the first GetObjectsWithTags + // measurement. Small enough to be safe on any machine regardless of + // tag-map size, large enough to make meaningful progress per + // FollowReferences call. static constexpr int GOTW_INITIAL_BATCH_SIZE = 64; - // AIMD bounds for the adaptive batch: MAX bounds JNI local refs per call, MIN keeps each call - // from resolving a single object. + // AIMD bounds for the adaptive batch. The cap bounds JNI local refs + // (resolved objects + holder array) per call; the floor keeps a + // degenerate tiny batch from making each FollowReferences call + // resolve a single object (observed live: batch=2 collapsed BFS + // throughput ~20x while per-call cost stayed ~20ms). static constexpr size_t GOTW_MAX_BATCH = 512; static constexpr size_t GOTW_MIN_BATCH = 8; - // Search lifecycle state. _search_started distinguishes a search's first pass (seed - // FollowReferences from the heap roots) from a resumed pass (expand the persisted frontier, see - // expandFrontier()) - runPass() below. + // Search lifecycle state. _search_started distinguishes a search's first + // pass (seed FollowReferences from the heap roots) from a resumed pass + // (expand the persisted frontier, see expandFrontier()) - runPass() below. + // _search_state starts RUNNING and only ever moves forward (RUNNING -> + // COMPLETED or RUNNING -> ABANDONED, never back) - see runPass()'s comment + // for the exact conditions. Both fields are written only by runPass(), + // called from the single agent-owned BFS thread, but are read cross-thread + // by searchState()/buildAbandonedEvent() (called from Profiler::dump(), + // e.g. profiler.cpp's JFR-flush path) - so, like _gc_start_epoch/ + // _gc_finish_epoch above, they are volatile and accessed via load()/ + // store() rather than a plain load/store the compiler could reorder or + // cache across threads. bool _search_started; volatile u8 _search_state; - // True once releaseSearchTags() has confirmed every live tag this search owned was actually - // cleared (or there were none) - see that method's own comment for why a GetObjectsWithTags() - // failure must NOT be treated as "released". + // True once releaseSearchTags() has confirmed every live tag this search + // owned was actually cleared (or there were none) - see that method's own + // comment for why a GetObjectsWithTags() failure must NOT be treated as + // "released". Starts true (nothing to release for a not-yet-run search); + // set false the moment a search reaches a terminal state and is only ever + // reset back to true once releaseSearchTags() itself confirms success - + // possibly across several retried runPass() calls first, see runPass()'s + // terminal-state branch. shouldRunPass() refuses to restartSearch() while + // this is false, so _next_tag/the frontier table are never reset out from + // under a search whose tags might still be live. Written and read only + // from the single BFS thread (runPass()/shouldRunPass()), like + // _search_started above, so no volatile/load()/store() is needed. bool _tags_released; - // Whether threadLoop()'s urgency ramp currently holds the multiplied _budget (see the urgency - // block there and _configured_budget's own comment). + // Whether threadLoop()'s urgency ramp currently holds the multiplied + // _budget (see the urgency block there and _configured_budget's own + // comment). Single-threaded (BFS thread), like _tags_released above. bool _urgency_budget_boosted; - // Hysteresis state behind isUrgent(), which used to be a bare `secondsToOOM() < - // OOM_URGENT_THRESHOLD_S` comparison. + // Hysteresis state behind isUrgent(), which used to be a bare + // `secondsToOOM() < OOM_URGENT_THRESHOLD_S` comparison. That estimate is + // computed from a short ring of heap deltas, so it swings by orders of + // magnitude between consecutive observations of the very same steadily + // growing heap (observed in one run: 128s, then 52769s, then back). + // _urgent_latched is set the first time the projection drops below + // OOM_URGENT_THRESHOLD_S and only cleared once it has stayed at or above + // OOM_URGENT_RELEASE_S (or gone unknown, i.e. negative) for + // URGENT_RELEASE_CONSECUTIVE consecutive observations, counted by + // _urgent_release_ticks. + // + // _urgent_search_spent makes each urgency episode authorize exactly one + // search. hasLeakSignal()'s urgency shortcut bypasses the per-klass + // hysteresis gate, so without this every terminal search reaching + // shouldRunPass()'s restart branch while still urgent immediately called + // restartSearch() again - which discards the frontier table and the + // leak-signature/parent-fanout accumulators (see restartSearch()), so the + // rotation heuristics that need several passes to converge were wiped + // before they ever could. Set when a search is started under a latched + // urgency, cleared together with the latch. The per-klass leak-candidate + // half of hasLeakSignal() is untouched and can still authorize restarts + // during an episode. + // + // All three are written and read only from the single BFS thread, but + // mutable because the latch is maintained inside isUrgent() const. mutable bool _urgent_latched; mutable int _urgent_release_ticks; mutable bool _urgent_search_spent; - // Set (once) at the same point runPass() moves _search_state to ABANDONED - see - // SearchAbandonReason's own comment for why this exists and buildAbandonedEvent()/abandonReason() - // below for how it is read. + // Set (once) at the same point runPass() moves _search_state to ABANDONED - + // see SearchAbandonReason's own comment for why this exists and + // buildAbandonedEvent()/abandonReason() below for how it is read. Same + // cross-thread read pattern as _search_state above. volatile u8 _abandon_reason; - // Wall-clock timestamp (OS::nanotime()) of the search's first pass - baseline for the TTL cutoff - // above. + // Wall-clock timestamp (OS::nanotime()) of the search's first pass - + // baseline for the TTL cutoff above. Set once, in runPass(), the first + // time _search_started flips true; read cross-thread by + // buildAbandonedEvent() (elapsed-time calculation), so volatile/load()- + // accessed like _search_state above. volatile u64 _search_start_ns; - // Tags currently in FrontierEntryState::FRONTIER (admitted but not yet expanded), in admission - // order. + // Tags currently in FrontierEntryState::FRONTIER (admitted but not yet + // expanded), in admission order. Pushed by heapReferenceCallback() at the + // moment it admits a tag (both for the first pass's root-seeded walk and + // for expandFrontier()'s own per-node FollowReferences calls, since both + // share that one callback), popped by expandFrontier()/ + // markAllFrontierExpanded() as entries are expanded. This replaces a + // former O(range) scan over every tag between a cursor and the frontier's + // current size just to filter down to the FRONTIER-state subset - a scan + // whose cost was proportional to everything admitted since the cursor + // last advanced, not to what was actually pending, so a pass immediately + // following a large one-shot admission (e.g. a restart's first pass) paid + // for the whole batch just to discover a handful of genuinely pending + // entries. Only ever touched from the single BFS thread that runs + // heapReferenceCallback()/expandFrontier(), so no locking is needed. std::deque _pending_expand; - // Fast-lane counterpart to _pending_expand above: entries admitted while re-walking a - // rotation-selected (already-EXPANDED) parent go here instead, and expandFrontier() drains this - // queue ahead of the ordinary one. + // Fast-lane counterpart to _pending_expand above: entries admitted while + // re-walking a rotation-selected (already-EXPANDED) parent go here + // instead, and expandFrontier() drains this queue ahead of the ordinary + // one. Without this, a mutable field re-observed via rotation (e.g. + // HashMap.table after a resize) admits a fresh child that then has to + // travel through however much of the ordinary backlog is still ahead of + // it - under a fast-growing leak that backlog can be tens of thousands of + // entries deep, so the re-admitted chain would never visibly progress + // within any reasonable search window. Same single-BFS-thread-only + // access as _pending_expand, no locking needed. + // + // HARD CAP (PRIORITY_EXPAND_CAP): the rotation collectors enqueue up to + // STALE_EXPANDED_ROTATION_BUDGET+ROOT_KIND_ROTATION_BUDGET entries per + // pass, but each phase's wall-clock deadline admits only ~2-3 + // GetObjectsWithTags calls (~50-200 entries) of drain per pass, so an + // uncapped queue grows without bound - observed live: 39k->103k in 20 + // minutes while _pending_expand (the BFS frontier itself) was never + // drained once, because expandFrontier() drains this queue first. The + // cap bounds both the memory and isQueuedForRotation()'s linear scan; + // collectors and admitObject() skip pushing when full, which throttles + // rotation to whatever the drain can actually consume. + // + // expandFrontier() alternates batches between this queue and + // _pending_expand when both are non-empty (see its own comment) - the + // original priority-first drain is what let this queue starve the + // ordinary backlog above. std::deque _priority_expand; - // Which lane the NEXT expandFrontier() batch comes from when both lanes are non-empty (the - // alternation toggle). + // Which lane the NEXT expandFrontier() batch comes from when both lanes + // are non-empty (the alternation toggle). Deliberately a MEMBER, not a + // local: the phase deadlines bound a typical expandFrontier() invocation + // to ONE batch (a single GetObjectsWithTags costs ~25-30ms of a 50ms + // window at a ~240k-entry tag map), and a per-invocation local reset to + // "priority first" made priority win EVERY invocation - observed live on + // hotdog, the ordinary _pending_expand lane (109k entries) was never + // drained by a single batch while the priority lane livelocked on stale + // re-walks. Persisting the toggle across invocations makes the two + // phases of each pass (expand + rotation) drain alternating lanes. bool _expand_lane_prefer_priority = true; - // Upper bound on _priority_expand above. 1024 holds a few passes' worth of rotation selection - // (budgets sum to ~272/pass) so a truncating rotation phase still has work waiting next pass. + // Upper bound on _priority_expand above. 1024 holds a few passes' worth of + // rotation selection (budgets sum to ~272/pass) so a truncating rotation + // phase still has work waiting next pass. static constexpr size_t PRIORITY_EXPAND_CAP = 1024; - // O(1) membership index over _priority_expand, backing isQueuedForRotation(): the rotation - // collectors run that check for EVERY FrontierTable slot they visit (~199k EXPANDED entries on a - // large heap), and the original linear scan over the deque cost up to ~200M comparisons per - // rotation pass at the cap - observed prominently in profiles. + // O(1) membership index over _priority_expand, backing + // isQueuedForRotation(): the rotation collectors run that check for + // EVERY FrontierTable slot they visit (~199k EXPANDED entries on a + // large heap), and the original linear scan over the deque cost up to + // ~200M comparisons per rotation pass at the cap - observed prominently + // in profiles. Fixed-capacity open addressing with no allocation after + // construction: PRIORITY_EXPAND_CAP (1024) live entries in a + // 2*PRIORITY_EXPAND_SLOT_SHIFT-power-of-two slot table at <=0.5 load + // factor, linear probing over Fibonacci-hashed tags (frontier tags are + // near-sequential, so a plain (tag % slots) index would cluster). + // Deletion needs NO tombstones: expandFrontier() pops a batch off the + // deque's front and rebuildFrom() re-derives the index from the deque's + // remaining contents - a full rebuild is <=1024 inserts, a few + // microseconds, against the ~20ms GetObjectsWithTags call the same + // batch already paid. The deque remains the drain-order source of + // truth; this index only answers membership. All mutation happens on + // the engine thread under the _engine_lock mutex (pushes in the + // collectors/admitObject()/requeueChainRootForRotation(), pops in + // expandFrontier()/markAllFrontierExpanded(), clears in + // startSearch()/restartSearch()), so plain non-atomic access is safe. class PriorityExpandSet { private: - // 2^11 == 2 * PRIORITY_EXPAND_CAP == 2048 slots. The shift below derives from it; keep both in - // sync. + // 2^11 == 2 * PRIORITY_EXPAND_CAP == 2048 slots. The shift below + // derives from it; keep both in sync. static constexpr u64 SLOT_SHIFT = 11; static constexpr u64 SLOT_MASK = (1ULL << SLOT_SHIFT) - 1; - jlong _keys[1ULL << SLOT_SHIFT]; - u8 _used[1ULL << SLOT_SHIFT]; // 0 = empty, 1 = occupied + // In-class zero-initialization: both members are raw arrays with no + // constructor, and the two member instantiations in + // ReferenceChainTracker are plain declarations - without these + // initializers, isQueued()/insert() would probe UNINITIALIZED _used + // bytes until the first clear()/rebuildFrom() ran (a constructor- + // ordering-dependent misclassification, and a worse one: _keys is + // read on a _used[i] hit, so garbage keys could "match" a live tag). + // Value-init costs nothing (zero pages, BSS). + jlong _keys[1ULL << SLOT_SHIFT] = {}; + u8 _used[1ULL << SLOT_SHIFT] = {}; // 0 = empty, 1 = occupied static u64 mix(jlong tag) { - // Fibonacci hashing: spreads near-sequential integer tags evenly across the table's - // power-of-two slot space. + // Fibonacci hashing: spreads near-sequential integer tags evenly + // across the table's power-of-two slot space. return (u64)tag * 0x9E3779B97F4A7C15ULL; } @@ -383,6 +1486,15 @@ class ReferenceChainTracker { } // Idempotent: returns false if `tag` is already indexed. + // Occupancy-bounded: a completely full table (2048/2048) would make + // the linear-probe loop wrap forever - a livelock on the engine + // thread inside rotation collection or rebuildFrom(). The external + // invariant (deque <= PRIORITY_EXPAND_CAP by construction) normally + // prevents this, but the invariant is enforced only at push sites; + // the occupancy check makes insert() self-terminating even if a + // future call site breaks it (insert returns false - the tag is + // treated as not-queued, the same degradation a full deque push + // already accepts). bool insert(jlong tag) { u64 i = mix(tag) >> (64 - SLOT_SHIFT); while (_used[i]) { @@ -390,6 +1502,10 @@ class ReferenceChainTracker { return false; } i = (i + 1) & SLOT_MASK; + // Wrapped all 2048 slots without an empty one: table full. + if (i == (mix(tag) >> (64 - SLOT_SHIFT))) { + return false; + } } _used[i] = 1; _keys[i] = tag; @@ -400,7 +1516,10 @@ class ReferenceChainTracker { memset(_used, 0, sizeof(_used)); } - // Re-derives the index from the deque's CURRENT contents. + // Re-derives the index from the deque's CURRENT contents. Call after + // any pops so membership matches the queue exactly again; the + // deque's own size is the only bound needed here (the engine thread + // guarantees it stays <= PRIORITY_EXPAND_CAP by construction). template void rebuildFrom(const Deque &queue) { clear(); for (jlong tag : queue) { @@ -409,290 +1528,709 @@ class ReferenceChainTracker { } } _priority_expand_set; - // B' (find-anchor-holder-eviction / find-anchor-live-feed-design): a static holder richly - // referenced from the running graph is EXCLUDED from the static-anchor tier forever once its - // frontier entry is chain-attached (first admitted via a non-root path, or demoted by - // improveChain) - maybeUpgradeRootAttachedRootKind() refuses entries with parent_tag != 0 by - // design, so collectStaticFieldAnchorsForRotation() (parent_tag == 0 filter) can never select it. + // B' (find-anchor-holder-eviction / find-anchor-live-feed-design): a + // static holder richly referenced from the running graph is EXCLUDED + // from the static-anchor tier forever once its frontier entry is + // chain-attached (first admitted via a non-root path, or demoted by + // improveChain) - maybeUpgradeRootAttachedRootKind() refuses entries + // with parent_tag != 0 by design, so collectStaticFieldAnchorsForRotation() + // (parent_tag == 0 filter) can never select it. This FIFO is the live + // feed that repairs the hole: the static sweep's class->field edge onto + // such an entry (heapReferenceCallback's root-like-onto-already-admitted + // block, on the failed maybeUpgradeRootAttachedRootKind()) pushes the + // tag here, and runPassManualWalk() drains it into the same + // walkStaticFieldAnchors() batch BEHIND the collector's small + // root-attached cohort (pod round 10: a cap-pinned at-risk flood starving + // that cohort in ~60% of passes is what the reverse order caused) - + // anchor selection no longer depends on the entry's attribution shape + // for this population, so the eviction is structurally impossible. + // Feed semantics: pushes happen every sweep lap (the sweep re-proves + // the static edge each lap), drained entries are NOT re-pushed by the + // walk - an un-intercepted holder comes back via the next lap's push, + // which bounds steady-state occupancy to one lap's worth of at-risk + // holders instead of a permanent rotation cohort. Same engine-thread + // under-_engine_lock discipline as _priority_expand above (the push + // site runs inside the sweep's FollowReferences in runPassSerialized(), + // the drain in the same pass's rotation phase), so no locking. Deque + // node allocations are bounded by the cap; pushes inside the JVMTI + // callback allocate only when a new deque node is needed - bounded, + // amortized, and on the engine thread, never a signal context. + // Round 16: entries carry their klass_id so occupancy is countable per + // class (see _static_anchor_fifo_klass_counts below) - drain/requeue + // maintain the counts exactly without a frontier lookup, so a holder + // whose entry died between push and drain cannot leak a stale count. struct AtRiskAnchor { jlong tag; u32 klass_id; - // PriorityExpandSet::rebuildFrom() iterates `jlong tag : queue` - the implicit conversion keeps - // that template generic over both the plain-jlong _priority_expand deque and this pair deque. + // PriorityExpandSet::rebuildFrom() iterates `jlong tag : queue` - the + // implicit conversion keeps that template generic over both the + // plain-jlong _priority_expand deque and this pair deque. operator jlong() const { return tag; } }; std::deque _static_anchor_fifo; - // Membership index over _static_anchor_fifo (push-side dedupe, so one lap's repeated static edges - // onto the same chain-attached holder push it once) - a second instance of PriorityExpandSet, - // whose fixed 2048 slot table keeps the cap at PRIORITY_EXPAND_CAP (1024). + // Membership index over _static_anchor_fifo (push-side dedupe, so one + // lap's repeated static edges onto the same chain-attached holder push + // it once) - a second instance of PriorityExpandSet, whose fixed 2048 + // slot table keeps the cap at PRIORITY_EXPAND_CAP (1024). Full-FIFO + // pushes are dropped (the natural throttle: at-risk holders far beyond + // one lap's drain rate wait for the next lap's re-push, never lost). PriorityExpandSet _static_anchor_fifo_set; static constexpr size_t STATIC_ANCHOR_FIFO_CAP = PRIORITY_EXPAND_CAP; - // Per-class admission counts backing the per-klass cap on at-risk anchor pushes. + // Round 16 (pod round-15 measurement, ev-leaktag-onpod-round15-results): + // the B' repair was DEAD on the pod - the FIFO sat cap-pinned at 1024 + // because three classes flooded it (klass 1: 1396 pushes, klass 215: + // 1063, klass 1733: 988+), so the LEAK_BUFFER wrapper's own pushes + // (klass 28366) were dropped at the cap check and the wrapper never + // rode B' - the exact at-risk holder the lane exists for. Per-class + // occupancy, maintained exactly by push/drain/requeue (the klass rides + // in each AtRiskAnchor), so one class cannot dominate the lane: at + // STATIC_ANCHOR_ATRISK_PER_KLASS_CAP entries per class the 1024 cap + // necessarily holds >= 16 distinct classes, and a class at its quota + // dropping its 65th push is correct twice over - it is already + // represented (its oldest entry drains within a few passes at + // STATIC_ANCHOR_FIFO_DRAIN=16/pass), and the alternative measured on + // the pod was every other class's repair being locked out. Counters + // are erased at zero on drain so the map is bounded by the FIFO's own + // contents (<= 1024 distinct classes), not by the search lifetime. std::unordered_map _static_anchor_fifo_klass_counts; static constexpr u32 STATIC_ANCHOR_ATRISK_PER_KLASS_CAP = 64; - // Index of root-attached STATIC_FIELD/JNI_GLOBAL frontier entries, so - // collectStaticFieldAnchorsForRotation() iterates O(anchors) instead of scanning the full - // frontier table O(frontier_size). + // Index of root-attached STATIC_FIELD/JNI_GLOBAL frontier entries, + // so collectStaticFieldAnchorsForRotation() iterates O(anchors) instead + // of scanning the full frontier table O(frontier_size). An entry is + // added when it is first admitted root-attached with a durable root_kind + // (STATIC_FIELD or JNI_GLOBAL), or when maybeUpgradeRootAttachedRootKind() + // upgrades it to one of those. Cleared on restartSearch(). Engine thread + // only — all mutation sites run under _engine_lock or inside the BFS + // thread's own pass. std::vector _static_anchor_index; - // Parallel to _static_anchor_index: the OWN class tag of each anchor object (the class of the - // static field's VALUE, not the holder class). + // Parallel to _static_anchor_index: the OWN class tag of each anchor + // object (the class of the static field's VALUE, not the holder class). + // The selection tiers anchors by that class's shape (container vs other, + // see AnchorClassShape below) via a lookup into _class_shape_cache, so a + // later classification automatically upgrades an entry's tier without + // any index mutation. Cleared with the index on restartSearch(). std::vector _static_anchor_own_class_tags; - // Dedup companion for _static_anchor_index (O(1) membership; the population is ~28k on a real JVM - // - see addToStaticAnchorIndex()'s own comment). + // Dedup companion for _static_anchor_index (O(1) membership; the + // population is ~28k on a real JVM - see addToStaticAnchorIndex()'s own + // comment). Cleared with the index. std::unordered_set _static_anchor_index_tags; - // Shape of a class as an anchor candidate: does the class implement java/util/Collection or - // java/util/Map (directly or via superclasses/ interfaces)? + // Shape of a class as an anchor candidate: does the class implement + // java/util/Collection or java/util/Map (directly or via superclasses/ + // interfaces)? A leak holder is typically a container (the hotdog leak: + // Collections$SynchronizedRandomAccessList, a List) while the anchor + // population is dominated by non-containers (String/Class/boxed/ + // primitive-array/enum statics - round 13 measured ~28k anchors on the + // hotdog JVM vs ~4k walkable per search). Selection walks leak-tagged + // anchors first, then containers, then everything else cursor-fairly — + // a container anchor is thus covered within ceil(container_cohort / + // budget) passes of admission instead of within ceil(28k / budget) + // passes (= never, at ~190-pass search lifetimes). enum class AnchorClassShape : u8 { UNKNOWN = 0, CONTAINER = 1, NON_CONTAINER = 2 }; - // class tag -> AnchorClassShape, process-lifetime (class tags are never reused - the shared - // class-tag allocator is deliberately not reset by restartSearch()). + // class tag -> AnchorClassShape, process-lifetime (class tags are never + // reused - the shared class-tag allocator is deliberately not reset by + // restartSearch()). Filled lazily by reconcileAnchorClassShapes(): an + // anchor's own class is classified on first need, never re-classified. + // Engine thread only. Entries are never evicted: bounded by loaded-class + // count (~34k on hotdog), u8 values. std::unordered_map _class_shape_cache; - // java/util/Collection and java/util/Map class tags, resolved once lazily by - // resolveContainerInterfaceTags() (0 = not yet resolved; a resolved value is NEGATIVE - class - // tags are a negative namespace, see nextClassTag()'s own comment). + // java/util/Collection and java/util/Map class tags, resolved once + // lazily by resolveContainerInterfaceTags() (0 = not yet resolved; a + // resolved value is NEGATIVE - class tags are a negative namespace, + // see nextClassTag()'s own comment). Class tags are stable for the + // JVM's lifetime, so these are safe to cache across searches. Engine + // thread only. jlong _collection_iface_class_tag = 0; jlong _map_iface_class_tag = 0; - // Fair-rotation cursors (index POSITIONS, not tags) for the two cursor-fair tiers of - // collectStaticFieldAnchorsForRotation(): _anchor_container_cursor for container-shaped anchors, - // _anchor_other_cursor for everything else. + // Fair-rotation cursors (index POSITIONS, not tags) for the two + // cursor-fair tiers of collectStaticFieldAnchorsForRotation(): + // _anchor_container_cursor for container-shaped anchors, + // _anchor_other_cursor for everything else. Leak-tagged anchors are + // always selected (they are rare) and need no cursor. Wrapping + // semantics: the next selection for a tier scans for entries at + // positions >= the cursor, stops at the lap end (no within-call wrap), + // and leaves the cursor just past the last consumed position. size_t _anchor_container_cursor = 0; size_t _anchor_other_cursor = 0; - // FIFO of newly admitted anchors awaiting their first walk. + // Fresh-admission lane (round 15): FIFO of anchor tags that have not + // yet had their ONE first-look walk priority. addToStaticAnchorIndex() + // appends here alongside the index (the two append together, so the + // queue is always an ordered suffix window of the index); the collector + // drains it each call - keeping container-shaped or not-yet-classified + // entries up to the walk budget, and DROPPING everything else out of + // the queue (a dropped anchor keeps its index position and is owned by + // the fair tiers from then on - it got its one first look and was + // outranked, not lost). Why a queue and not a "positions since mark" + // watermark: the mark form reorders selection when state leaks between + // units/tests (a stale mark demotes an early anchor behind a later one + // for no reason - caught by StaticAnchorRotation* in test-order runs); + // the queue is per-anchor, so each anchor's first look is exactly once, + // in admission order, regardless of what any earlier search/test left + // behind. Rationale for the lane itself: admission order = sweep order + // = loaded-class order, so a leak holder held by a late-loaded class + // (the hotdog wrapper: holder class at sweep index 24627 of 33270) + // admits at the index TAIL - the very END of the fair container tier's + // lap - and the measured search lifetime (44-75 passes, round 15) is + // shorter than ceil(container_cohort/budget) (1633/16 = 102), so + // fair-only coverage never reaches it; the fresh lane walks it within a + // pass or two of admission instead. Bounded by STATIC_ANCHOR_FRESH_CAP + // (overflow drops the OLDEST fresh chances - they fall back to the fair + // tiers, never lost). Cleared with the index on restartSearch(). std::deque _static_anchor_fresh_queue; - // Cap for _static_anchor_fresh_queue. The queue normally holds only ~one pass of sweep admits - // (admits happen in passes, the collector drains every pass, and the drain DROPS everything it - // does not keep, so the queue empties each call); the cap only guards a pathological burst (e.g. - // a pass admitting thousands) from growing it unbounded in native memory. + // Cap for _static_anchor_fresh_queue. The queue normally holds only + // ~one pass of sweep admits (admits happen in passes, the collector + // drains every pass, and the drain DROPS everything it does not keep, + // so the queue empties each call); the cap only guards a pathological + // burst (e.g. a pass admitting thousands) from growing it unbounded in + // native memory. static constexpr size_t STATIC_ANCHOR_FRESH_CAP = 1024; - // java/lang/Object jclass cache for expandFrontier()'s and admitStaticFieldRoots()'s holder-array - // element type (referenceChains.cpp) - resolved once via FindClass()+NewGlobalRef() and reused - // for the tracker's lifetime. + // java/lang/Object jclass cache for expandFrontier()'s and + // admitStaticFieldRoots()'s holder-array element type (referenceChains.cpp) + // - resolved once via FindClass()+NewGlobalRef() and reused for the + // tracker's lifetime. MUST be a GLOBAL ref, not a local one: callers + // include JNI-entered test seams (runReferenceChainPass0), and a local + // ref is freed the moment its creating JNI invocation returns to Java - + // caching one across invocations crashed in NewObjectArray() on the + // second pass (observed). A global ref is also valid across the BFS + // thread's detach/attach cycles, so no JNIEnv* keying or detach-time + // invalidation is needed. Never freed: java/lang/Object is a bootstrap + // class (never unloaded) and this tracker is a process-lifetime + // singleton, so the single ref is reclaimed with the JVM. jclass _cached_object_class = nullptr; - // Rotation cursor for collectStaleRootKindEntriesForRotation(): 1-based tag to resume scanning - // from on the next call, so consecutive calls sweep forward through the table instead of always - // re-examining the same low-tag entries first. + // Rotation cursor for collectStaleRootKindEntriesForRotation(): 1-based + // tag to resume scanning from on the next call, so + // consecutive calls sweep forward through the table instead of always + // re-examining the same low-tag entries first. Wraps back to 1 once it + // reaches _frontier->size(). Persisted across passes (not per-search-reset + // by ReferenceChainsTestAccessor::reset(), same as _next_tag is not reset + // by restartSearch() logic elsewhere) since a stale cursor value only ever + // costs one wasted scan step before self-correcting, never a correctness + // problem. jlong _root_kind_rotation_cursor; // Same role as _root_kind_rotation_cursor above, but for - // collectStaleExpandedEntriesForRotation(): without its own persistent cursor, that sweep always - // restarted from tag 1 on every call, so a frontier table holding >= - // STALE_EXPANDED_ROTATION_BUDGET low-tag entries that stay EXPANDED forever (long-lived - // infrastructure objects) filled its entire per-pass cap from that population alone, every pass, - // permanently starving any EXPANDED entry with a higher tag (e.g. a static field's collection, - // admitted only once its class loads well after startup) of ever being re-queued. + // collectStaleExpandedEntriesForRotation(): without its own persistent + // cursor, that sweep always restarted from tag 1 on every call, so a + // frontier table holding >= STALE_EXPANDED_ROTATION_BUDGET low-tag entries + // that stay EXPANDED forever (long-lived infrastructure objects) filled + // its entire per-pass cap from that population alone, every pass, + // permanently starving any EXPANDED entry with a higher tag (e.g. a + // static field's collection, admitted only once its class loads well + // after startup) of ever being re-queued. jlong _stale_expanded_rotation_cursor; - // Per-pass cap on how many transient-root_kind entries collectStaleRootKindEntriesForRotation() - // selects - round, provisional like this subsystem's other unbenchmarked constants (see e.g. - // MIN_EFFECTIVE_BUDGET's own comment): small enough that a pass dominated by rotation work never - // meaningfully competes with genuinely new discoveries for the same pass's budget, large enough - // that a search with a modest number of transient roots converges to durable attribution within a - // handful of passes rather than needing hundreds. + // Per-pass cap on how many transient-root_kind entries + // collectStaleRootKindEntriesForRotation() selects - round, provisional + // like this subsystem's other unbenchmarked constants (see e.g. + // MIN_EFFECTIVE_BUDGET's own comment): small enough that a pass dominated + // by rotation work never meaningfully competes with genuinely new + // discoveries for the same pass's budget, large enough that a search with + // a modest number of transient roots converges to durable attribution + // within a handful of passes rather than needing hundreds. static constexpr int ROOT_KIND_ROTATION_BUDGET = 16; - // Per-pass cap on how many EXPANDED entries collectStaleExpandedEntriesForRotation() re-queues - // for expansion, uniformly across the WHOLE frontier table regardless of lineage. + // Per-pass cap on how many EXPANDED entries + // collectStaleExpandedEntriesForRotation() re-queues for expansion, + // uniformly across the WHOLE frontier table regardless of lineage. This is + // the low-priority fallback tier of the rotation design (see + // collectLeakAccumulationCandidatesForRotation()'s own comment for the + // targeted tier): coverage of the frontier table is only guaranteed within + // ceil(table_size / this) passes, which can be far longer than any one + // search realistically survives before completing/restarting on a large + // table - two earlier versions of this code tried to compensate by scaling + // this cap with table size, then by adding a structural (depth + root- + // durability + class-shape) priority tier, but both were solving the wrong + // problem (see git history): no purely structural property can distinguish + // the one specific container that is actually leaking from the thousands + // of ordinary ones a real classpath contains. collectLeakAccumulationCandidatesForRotation() + // instead targets that population directly, using LivenessTracker's own + // growth signal plus fanout-of-a-flagged-klass, at a small fixed budget + // regardless of table size. This constant stays flat because the rest of + // the table (everything NOT tied to a currently-flagged klass) genuinely + // doesn't need a faster guarantee - eventual coverage is enough. static constexpr int STALE_EXPANDED_ROTATION_BUDGET = 256; // Candidate-scoped reach (see descendFromAnchor()/walkCandidateThreadLocals()/ - // walkStaticFieldAnchors()'s own comments): how many hops BELOW a descend walk's anchor the walk - // may admit. + // walkStaticFieldAnchors()'s own comments): how many hops BELOW a descend + // walk's anchor the walk may admit. A static Map -> table[] -> Entry -> + // leaked chunk is 3-4 hops below its root-attached holder; a Thread -> + // ThreadLocalMap -> table[] -> Entry -> value -> holder -> chunk is 5-6 + // below the Thread object. Raised 6 -> 16 after pod round 7: with both + // prongs live, interception stayed zero while the static-anchor walks + // admitted whole executor-task subgraphs (3000+ edges in single walks) + // - the holder interior is reachable but the tagged chunks sit one or + // more hops PAST the 6-hop cap (the static-ExecutorService -> queue -> + // task -> accumulator -> list -> chunk shape is ~7 deep). The walk is + // deadline-bounded per slice either way, so the cap's cost model does + // not change - a deeper cap just lets each bounded walk cover the whole + // holder interior instead of stopping mid-way, and already-admitted + // entries are skipped so repeated passes march deeper each time. static constexpr int DESCENT_HOPS = 16; - // Per-pass cap on how many candidate threads walkCandidateThreadLocals() descend-walks. + // Per-pass cap on how many candidate threads walkCandidateThreadLocals() + // descend-walks. Each anchor walk is a whole bounded FollowReferences + // call, so unlike queue-push rotation tiers, this cap is the real cost + // control next to the deadline; a per-pass rotation of 4 gives every + // qualifying tid a turn within a few passes (qualifying tid sets are + // small - bounded by MAX_QUALIFYING_TIDS per candidate, and typically 1-2). static constexpr int THREAD_WALK_MAX_ANCHORS = 4; - // Per-pass cap on how many root-attached static holders walkStaticFieldAnchors() resolve + - // descend-walk. + // Per-pass cap on how many root-attached static holders + // walkStaticFieldAnchors() resolve + descend-walk. Same "the cap IS the + // cost control" reasoning as THREAD_WALK_MAX_ANCHORS (each anchor is a + // bounded FollowReferences call); the tiered cursors in + // collectStaticFieldAnchorsForRotation() guarantee coverage of each + // tier within ceil(tier / this) passes - in particular the container + // tier (the likely leak-holder cohort) within ceil(container_cohort / + // this) passes of admission, regardless of how large the full anchor + // population is. Round 15 raised 16 -> 32: the measured fresh-container + // admit rate on hotdog (~10-37/pass depending on the sweep's class + // band) exceeded the old cap, which would have made the fresh-priority + // lane a growing backlog - same tail-starvation shape one level down. + // STW safety does NOT depend on this number: walkStaticFieldAnchors() + // stops at the per-pass deadline and requeues whatever it could not + // walk (TruncatedAnchorWalkRequeuesUnwalkedFifoTags covers the requeue + // path), so a larger selection can only spend selection-scan time + // (O(anchors), no safepoint), never extend the pause. Observed on-pod: + // the deadline already truncates 16-selection passes ("walked 6-16 of + // selected 20", round 10) - selection size and walked size are + // decoupled. static constexpr int STATIC_ANCHOR_ROTATION_BUDGET = 32; - // Per-pass cap on how many DISTINCT anchor classes reconcileAnchorClassShapes() classifies (one - // GetObjectsWithTags call for the batch + a depth-bounded interface walk per class). + // Per-pass cap on how many DISTINCT anchor classes + // reconcileAnchorClassShapes() classifies (one GetObjectsWithTags call + // for the batch + a depth-bounded interface walk per class). Bounds the + // first-lap classification of a ~34k-class JVM to ~270 passes worst case + // (in practice far fewer: only classes that actually own admitted + // anchors need classifying), after which the cache is warm for the JVM's + // lifetime. Same "the cap IS the cost control" reasoning as the budgets + // above. static constexpr int ANCHOR_SHAPE_RECONCILE_BUDGET = 128; - // Per-pass cap on how many AT-RISK static holders (frontier entries with parent_tag != 0 - the - // find-anchor-holder-eviction population) drainStaticAnchorFifo() pops for the same - // walkStaticFieldAnchors() batch. + // Per-pass cap on how many AT-RISK static holders (frontier entries + // with parent_tag != 0 - the find-anchor-holder-eviction population) + // drainStaticAnchorFifo() pops for the same walkStaticFieldAnchors() + // batch. Deliberately larger than STATIC_ANCHOR_ROTATION_BUDGET: the + // whole batch resolves in ONE GetObjectsWithTags call whose cost is + // dominated by the O(tag_map) scan floor, so a larger batch is nearly + // free per anchor; each anchor's own walk still draws down the same + // rotation budget, and truncation re-queues un-walked entries + // (requeueStaticAnchorFifoFront()), so a big batch costs only the + // per-entry GOTW bookkeeping, not extra STW. static constexpr int STATIC_ANCHOR_FIFO_DRAIN = 16; - // (Removed: the old single wrapping-index-cursor selection was replaced by tiered selection with - // per-tier cursors — see collectStaticFieldAnchorsForRotation()'s own comment and + // (Removed: the old single wrapping-index-cursor selection was replaced + // by tiered selection with per-tier cursors — see + // collectStaticFieldAnchorsForRotation()'s own comment and // _anchor_container_cursor/_anchor_other_cursor.) - // Cursor over the flattened (slot, tid) enumeration of _candidate_qualifying_tids above, so - // walkCandidateThreadLocals()'s THREAD_WALK_MAX_ANCHORS-per-pass cap rotates fairly instead of - // always walking the same first candidates' tids. + // Cursor over the flattened (slot, tid) enumeration of + // _candidate_qualifying_tids above, so walkCandidateThreadLocals()'s + // THREAD_WALK_MAX_ANCHORS-per-pass cap rotates fairly instead of always + // walking the same first candidates' tids. int _thread_walk_anchor_cursor; - // Per-pass cap on how many EXPANDED entries collectLeakAccumulationCandidatesForRotation() - // re-queues - see that method's own comment. + // Per-pass cap on how many EXPANDED entries + // collectLeakAccumulationCandidatesForRotation() re-queues - see that + // method's own comment. Small and fixed like the other two tiers' + // budgets: the population it selects from (_leak_parent_fanout) is itself + // already small, bounded by how many distinct parent objects have ever + // been observed holding an instance of one of the currently-watched leak + // klass_ids (see _watched_leak_klass_ids), not by total table size. static constexpr int LEAK_ACCUMULATION_ROTATION_BUDGET = 16; - // Snapshot of gcFinishEpoch() as of the end of the last pass. + // Snapshot of gcFinishEpoch() as of the end of the last pass. Written only + // by runPass(), read only by shouldRunPass() - both always called from the + // same thread (the BFS thread once wired up, or directly by a caller/test + // standing in for it), so no locking is needed. u64 _last_pass_gc_finish_epoch; - // OS::nanotime() as of the end of the last pass. Written by runPass() on the BFS thread; also - // read cross-thread by buildAbandonedEvent()'s elapsed-time calculation, so - // volatile/load()-accessed like _search_state above. + // OS::nanotime() as of the end of the last pass. Written by runPass() on + // the BFS thread; also read cross-thread by buildAbandonedEvent()'s + // elapsed-time calculation, so volatile/load()-accessed like + // _search_state above. volatile u64 _last_pass_ns; - // Total passes run this search. Written by runPass() on the BFS thread; read cross-thread by - // passesRun()/buildAbandonedEvent(), so volatile/ load()-accessed like _search_state above. + // Total passes run this search. Written by runPass() on the BFS thread; + // read cross-thread by passesRun()/buildAbandonedEvent(), so volatile/ + // load()-accessed like _search_state above. volatile int _passes_run; - // Resolved reference chains, keyed by the leak-candidate klass_id pollWatchedTargets() - // reconstructed each one for. + // Resolved reference chains, keyed by the frontier tag of the individual + // discovered instance the chain was reconstructed for (see the comment on + // _resolved_chains below for why per-instance, not per-klass_id). An + // entry lives here for as long as LivenessTracker can still resolve a + // representative for its klass_id: pollWatchedTargets() prunes it the + // moment the representative stops resolving (collected, or LRU-evicted + // from the population table). + // Every Profiler::dump() re-emits the whole cache (drainPendingChainEvents() + // snapshots without clearing), so a datadog.ReferenceChain lands in every + // JFR chunk the sample survives into - mirroring how LivenessTracker + // re-emits its live-object samples on each flush, rather than the emit-once + // model where a chain reached only the single transient dump that drained + // it. The frontier tag keying does NOT survive a search restart (which + // resets FrontierTable tags and can reissue the same numeric tag to a + // different object) - that is what CachedChain's source_search_ns stamp + // is for: a poll after a restart compares the stamp against the new + // search's _search_start_ns and rebuilds the chain instead of trusting a + // tag value the reset has since reassigned to a different object. struct CachedChain { ReferenceChainEvent event; jlong source_tag; u64 source_search_ns; }; - // Bounded by the number of discovered instances across all candidate slots: - // MAX_LEAK_CANDIDATES_FROM_LT * MAX_DISCOVERED_INSTANCES_PER_CLASS = 5 * 8 = 40, plus up to 5 - // canary chains. + // Bounded by the number of discovered instances across all candidate + // slots: MAX_LEAK_CANDIDATES_FROM_LT * MAX_DISCOVERED_INSTANCES_PER_CLASS + // = 5 * 8 = 40, plus up to 5 canary chains. 128 gives headroom for + // chains from the normal (non-canary) path too. static constexpr int MAX_RESOLVED_CHAINS = 128; // Keyed by frontier tag (individual instance identity), NOT by klass_id. + // For common classes like [B or [Ljava/lang/String; there may be many + // live instances with different reference chains — the first chain found + // may be noise (a shallow JNI-local instance), while the actual leak is + // a deep static-field-held instance. Caching per-instance ensures all + // chains are emitted to JFR and the profiling backend can aggregate them + // by class. Chains expire when the search restarts (frontier is wiped, + // all tags become invalid). std::unordered_map _resolved_chains; SpinLock _resolved_chains_lock; // Abandoned-search events awaiting Profiler::dump() (profiler.cpp). + // Populated synchronously by runPass() (referenceChains.cpp), at the exact + // point it writes SearchState::ABANDONED, by building a + // ReferenceChainAbandonedEvent from the fields that are only valid right + // then. Reading those fields lazily later (buildAbandonedEvent(), a live + // re-read of _search_state/_abandon_reason/etc.) does not work in + // production: shouldRunPass() (this class's own header comment) can call + // restartSearch() as soon as the very next BFS-thread loop iteration + // (~1s later), which both flips _search_state back to RUNNING and clears + // _abandon_reason/_search_start_ns/_passes_run - while Profiler::dump() + // only samples this tracker's state on JFR chunk rotation, an independent + // and much slower clock (tens of seconds). A dump landing outside that + // ~1s window previously saw nothing to report at all, silently dropping + // every abandoned-search event. Queueing the fully-built event at the + // moment of abandon removes the race entirely: dump() drains whatever + // accumulated since its last call, however long that took. + // + // Bounded like _resolved_chains above: an abandon queued between two + // dumps should be rare (dump cadence is normally much shorter than how + // long a search takes to get stuck), so hitting the cap and dropping + // (counted via REFERENCE_CHAIN_EVENTS_DROPPED, "no silent truncation") + // signals abandons happening far faster than normal, worth surfacing + // rather than growing without bound. static constexpr int MAX_PENDING_ABANDONED_EVENTS = 16; std::vector _pending_abandoned_events; SpinLock _pending_abandoned_events_lock; - // leaky bucket over the wall-clock cost of past searches, gating how soon a *restarted* search - // may take its first pass - see PainBudget's own comment (painBudget.h) and canAffordNewSearch() - // below. + // Deferred resolved-chain invalidations from the heap-walk callback + // (deferResolvedChainInvalidation(), drained by + // drainPendingChainInvalidations()). Small: each callback-side + // invalidation needs an improved/re-parented/upgraded frontier entry, and + // the list is drained after every pass. Capped defensively at + // MAX_PENDING_CHAIN_INVALIDATIONS (= 4 * MAX_RESOLVED_CHAINS) - overflow + // drops the deferral (the stale cached chain then lives until the search + // restart wipes the cache, the same lifetime the cache itself has). + static constexpr int MAX_PENDING_CHAIN_INVALIDATIONS = 4 * MAX_RESOLVED_CHAINS; + std::vector _pending_chain_invalidations; + SpinLock _pending_chain_invalidations_lock; + + // Search restart (this class's own header comment): leaky bucket over the + // wall-clock cost of past searches, gating how soon a *restarted* search + // may take its first pass - see PainBudget's own comment (painBudget.h) + // and canAffordNewSearch() below. _search_pain_ms accumulates the current + // search's own cost (each pass's pass_wall_ticks, converted to ms) as it + // runs; restartSearch() spends the total into _safepoint_pain_budget and zeroes this + // back out for the next search. Constructed with the configured refill + // rate in start(), mirroring _pause_pid's own placeholder-then-reconstruct + // pattern. PainBudget _safepoint_pain_budget; - // Cached refill rate from start(), reused by resetSearchStateForTest() so a test reset rebuilds - // the budget with the same rate. + // Cached refill rate from start(), reused by resetSearchStateForTest() + // so a test reset rebuilds the budget with the same rate. double _pain_budget_refill_rate = 0.0; u64 _search_pain_ms; - // Non-safepoint CPU-time pain budget: gates shouldRunPass() independently of both - // _safepoint_pain_budget above (which only cools down *restarts*, spent once per finished search) - // and _pause_pid's per-pass signal (updatePacing(), now fed only the genuine in-safepoint portion - // of each pass - see runPass()'s own comment). + // Non-safepoint CPU-time pain budget: gates shouldRunPass() independently + // of both _safepoint_pain_budget above (which only cools down *restarts*, spent once + // per finished search) and _pause_pid's per-pass signal (updatePacing(), + // now fed only the genuine in-safepoint portion of each pass - see + // runPass()'s own comment). Spent every pass, root-enum or not, with the + // wall-clock cost of runPassManualWalk() *outside* its actual JVMTI calls + // - root/stack-ref enumeration dispatch, frontier-table admission, + // rotation-candidate collection - none of which the pause-time PID has any + // visibility into. Seeded from the same configured refill rate as + // _safepoint_pain_budget (_reference_chains_pain_budget_percent) rather than a + // separate knob - one operator-facing "how much background cost is + // acceptable" percentage covers both leaky buckets. PainBudget _cpu_pain_budget; - // The cache above is mutated on this tracker's own BFS scheduling thread (pollWatchedTargets()) - // and read on whatever thread calls Profiler::dump() (drainPendingChainEvents()); - // _resolved_chains_lock (declared with the cache) is the only synchronization between them. - - // Fallback cadence between passes. + // The cache above is mutated on this tracker's own BFS scheduling thread + // (pollWatchedTargets()) and read on whatever thread calls Profiler::dump() + // (drainPendingChainEvents()); _resolved_chains_lock (declared with the + // cache) is the only synchronization between them. The write itself is + // still deferred to the dump() thread - the profiler-side writer can + // block up to ~50ms per event under _locks[] contention, + // which must never delay the next scheduled BFS pass - exactly mirroring + // how buildAbandonedEvent()'s output is deferred to that same call site + // rather than written eagerly. + + // Fallback cadence for shouldRunPass()'s cadence trigger (design doc's + // Triggering section / Open Question 5). Provisional default pending + // empirical tuning - not benchmark-derived: a round one-second value chosen + // only so an idle + // search still makes some progress between GC-triggered wakeups without + // polling so tightly that an idle tracker burns CPU. The pause-time pacing + // controller folds Open Question 5's cadence decision into updatePacing() + // below rather than solving it separately (design doc's explicit "one + // shared mechanism" framing): this constant now only serves as + // _effective_cadence_ns's starting value (start()) and as the unit + // MAX_EFFECTIVE_CADENCE_NS below scales from - shouldRunPass()/threadLoop() + // themselves compare against _effective_cadence_ns, not this constant + // directly, once a pass has run. static constexpr u64 PASS_CADENCE_NS = 1000000000ULL; // 1s - // Heap-wide time-to-OOM urgency threshold (LivenessTracker::secondsToOOM()) - hasLeakSignal() - // below forces a search to start immediately once the projection drops under this, rather than - // waiting for a klass to clear selectLeakCandidates()'s own per-klass ring-fill/hysteresis gate - // (KLASS_POPULATION_MIN_FILL_FOR_TREND plus LEAK_TREND_HYSTERESIS_BASE/ CORROBORATED epochs, - // livenessTracker.h). + // Heap-wide time-to-OOM urgency threshold (LivenessTracker::secondsToOOM()) + // - hasLeakSignal() below forces a search to start immediately once the + // projection drops under this, rather than waiting for a klass to clear + // selectLeakCandidates()'s own per-klass ring-fill/hysteresis gate + // (KLASS_POPULATION_MIN_FILL_FOR_TREND plus LEAK_TREND_HYSTERESIS_BASE/ + // CORROBORATED epochs, livenessTracker.h). Without this, an aggressive, + // heap-wide leak can OOM the process before any single klass ever clears + // that gate. Provisional, same status as every other un-benchmarked + // *_NS/*_S constant in this file: 5 minutes is chosen only to leave some + // margin for a BFS pass plus a JFR flush to actually run before the + // projected exhaustion, not measured against a real OOM race. static constexpr double OOM_URGENT_THRESHOLD_S = 300.0; // 5 minutes - // Release side of OOM_URGENT_THRESHOLD_S's hysteresis (see _urgent_latched). + // Release side of OOM_URGENT_THRESHOLD_S's hysteresis (see + // _urgent_latched). secondsToOOM() is derived from a short ring of heap + // deltas, so consecutive readings on the same monotonically growing heap + // routinely swing across OOM_URGENT_THRESHOLD_S in both directions - a + // single bare threshold comparison therefore flaps, and each flap back to + // "urgent" used to authorize a brand-new whole-heap search via + // hasLeakSignal(). Urgency is only released once the projection has stayed + // clear of this (deliberately higher) bar for URGENT_RELEASE_CONSECUTIVE + // consecutive observations. static constexpr double OOM_URGENT_RELEASE_S = 2 * OOM_URGENT_THRESHOLD_S; static constexpr int URGENT_RELEASE_CONSECUTIVE = 5; - // Horizon over which threadLoop() ramps the pause target and cadence toward their urgent ceilings - // as secondsToOOM() falls, once it reports a confirmed rising trend (see secondsToOOM()'s own - // NOT_RISING gate — a non-negative value already means real growth, not noise). + // Horizon over which threadLoop() ramps the pause target and cadence + // toward their urgent ceilings as secondsToOOM() falls, once it reports a + // confirmed rising trend (see secondsToOOM()'s own NOT_RISING gate — a + // non-negative value already means real growth, not noise). Profiles are + // flushed once per minute by default, so a search that only ramps up in + // the last OOM_URGENT_THRESHOLD_S (5 minutes) may not get enough elevated + // passes recorded in a JFR before the process dies. 30 minutes gives it + // ~30 recording rotations to actually land a chain. Separate from + // OOM_URGENT_THRESHOLD_S, which still gates hasLeakSignal()'s forced + // search start and runPass()'s TTL-abandonment suppression. static constexpr double OOM_RAMP_START_S = 1800.0; // 30 minutes - // Ceilings the pause target and cadence ramp toward as secondsToOOM() approaches zero within - // OOM_RAMP_START_S: the ramp is exponential (slow near OOM_RAMP_START_S out, aggressive near OOM) - // since the process is likely to die anyway and diagnostic data collected right before that is - // worth spending STW time and CPU on. + // Ceilings the pause target and cadence ramp toward as secondsToOOM() + // approaches zero within OOM_RAMP_START_S: the ramp is exponential (slow + // near OOM_RAMP_START_S out, aggressive near OOM) since the process is + // likely to die anyway and diagnostic data collected right before that is + // worth spending STW time and CPU on. TTL abandonment is suppressed + // whenever isUrgent() (see OOM_URGENT_THRESHOLD_S). static constexpr long URGENT_PAUSE_TARGET_MS = 100; // ceiling STW ms per pass static constexpr u64 URGENT_CADENCE_NS = 10000000ULL; // 10ms floor between passes - // True when LivenessTracker::secondsToOOM() projects exhaustion sooner - - // Auto-scaled default for _first_pass_budget when Arguments::_reference_chains_first_pass_budget - // is unset (0) - see _first_pass_budget's own comment for why plain _budget is the wrong - // fallback. + // Auto-scaled default for _first_pass_budget when + // Arguments::_reference_chains_first_pass_budget is unset (0) - see + // _first_pass_budget's own comment for why plain _budget is the wrong + // fallback. 50x is a round, provisional guess (same status as every other + // _reference_chains* constant here), picked to comfortably clear a cold + // JVM's root set without needing firstpassbudget spelled out explicitly for + // every reasonably-sized heap; the cap keeps a pathologically large + // _budget (e.g. an operator-supplied 100000) from ballooning the first + // pass's own one-shot cost unbounded. static constexpr int AUTO_FIRST_PASS_BUDGET_MULTIPLIER = 50; static constexpr int AUTO_FIRST_PASS_BUDGET_CAP = 200000; - // Minimum wall-clock gap between root/stack-ref enumeration attempts (runPassManualWalk()'s - // IterateOverReachableObjects call) after the search's own first pass. + // Minimum wall-clock gap between root/stack-ref enumeration attempts + // (runPassManualWalk()'s IterateOverReachableObjects call) after the + // search's own first pass. That call re-walks every live GC root and + // stack/JNI-local on every attempt regardless of budget (see its own + // comment) - a fixed tax independent of how much of it is new. Retrying it + // on every cheap steady-state tick (PASS_CADENCE_NS once relaxed) pays that + // tax far more often than it buys new admissions, which two live + // experiments confirmed nets LESS total progress than one large, + // infrequent attempt (each retry given _first_pass_budget-sized headroom - + // see _last_root_enum_ns's own comment). Seconds, not milliseconds: large + // enough that most ticks take the cheap expandFrontier()-only path, small + // enough that a ~20s search window still gets several independent attempts + // at whatever root JVMTI's enumeration order didn't reach the first time. + // Round, provisional guess, like this subsystem's other unbenchmarked + // constants. static constexpr u64 ROOT_ENUM_MIN_INTERVAL_NS = 2000000000ULL; // 2s - // Pause-time pacing controller: bounds and conversion constants for updatePacing()'s - // budget/cadence adjustment - see that method's own comment for the full mechanism. + // Pause-time pacing controller: bounds and conversion constants for + // updatePacing()'s budget/cadence adjustment - see that method's own + // comment for the full mechanism. Every value here is a round, provisional + // guess like every other _reference_chains* constant in this codebase + // (arguments.h's own DEFAULT_REFERENCE_CHAINS_* header comment sets the + // pattern) - a future benchmark plan is the intended path to replacing + // all of them with measured values, not a design decision made here. + // + // Floor updatePacing() will never shrink _effective_budget below (clamped + // further down to _budget itself when the configured budget is smaller + // than this floor - see updatePacing()). Not 0: a floor of 0 would let a + // single pathological pass shrink the search to "admit nothing, ever", + // stalling all progress instead of just slowing it. static constexpr int MIN_EFFECTIVE_BUDGET = 2000; - // Bounds for _effective_cadence_ns. The lower bound is not 0: threadLoop() sleeps for exactly - // this many nanoseconds each loop iteration (below), so a true 0 would busy-loop the BFS thread. + // Bounds for _effective_cadence_ns. The lower bound is not 0: threadLoop() + // sleeps for exactly this many nanoseconds each loop iteration (below), so + // a true 0 would busy-loop the BFS thread. The upper bound reuses + // PASS_CADENCE_NS (this field's own pre-pacing-controller baseline) as the unit for + // a round, provisional multiplier, so a search that is persistently over + // the pause-time ceiling still makes some progress rather than backing + // off indefinitely. static constexpr u64 MIN_EFFECTIVE_CADENCE_NS = 10000000ULL; // 10ms static constexpr u64 MAX_EFFECTIVE_CADENCE_NS = PASS_CADENCE_NS * 4; // 4s - // Conversion factor from "edges of budget signal updatePacing()'s clamp could not absorb" to a - // cadence adjustment in nanoseconds - the two are different units (edge count vs. + // Conversion factor from "edges of budget signal updatePacing()'s clamp + // could not absorb" to a cadence adjustment in nanoseconds - the two are + // different units (edge count vs. wall-clock time) with no natural + // exchange rate, so this is a round, provisional choice: large enough + // that a sustained, deeply-saturated overflow visibly moves the cadence + // within a handful of passes, small enough that a single borderline pass + // does not swing the whole cadence range at once. static constexpr u64 CADENCE_NS_PER_EDGE_OVERFLOW = 1000000ULL; // 1ms/edge - // Budget-borrowing (see _borrowed_budget's own comment): how many consecutive - // comfortably-under-target passes (BORROW_UNDER_TARGET_FRACTION) must be observed before - // updatePacing() starts growing _borrowed_budget at all. + // Budget-borrowing (see _borrowed_budget's own comment): how many + // consecutive comfortably-under-target passes (BORROW_UNDER_TARGET_FRACTION) + // must be observed before updatePacing() starts growing _borrowed_budget at + // all. Round, provisional like this subsystem's other unbenchmarked + // constants - large enough that a brief lull (e.g. one quiet pass right + // after a GC) cannot itself unlock extra headroom, small enough that a + // workload with a genuinely fast-growing frontier converges within a few + // seconds of passes rather than needing to wait out most of the search's + // own TTL just to start borrowing. static constexpr int BORROW_WARMUP_PASSES = 5; - // Budget-borrowing: a pass counts toward BORROW_WARMUP_PASSES/keeps _borrowed_budget only when - // pass_ms is at most this fraction of _pause_target_ms - deliberately stricter than merely "under - // the ceiling" (which the ordinary _effective_budget clamp already guarantees), so growth is - // gated on *comfortable* headroom, not on shaving the pass in just under the wire. + // Budget-borrowing: a pass counts toward BORROW_WARMUP_PASSES/keeps + // _borrowed_budget only when pass_ms is at most this fraction of + // _pause_target_ms - deliberately stricter than merely "under the + // ceiling" (which the ordinary _effective_budget clamp already + // guarantees), so growth is gated on *comfortable* headroom, not on + // shaving the pass in just under the wire. static constexpr double BORROW_UNDER_TARGET_FRACTION = 0.5; - // Budget-borrowing: hard cap on how far updatePacing() may grow (_budget + _borrowed_budget) - // above _budget alone - _borrowed_budget itself is clamped so the resulting ceiling never exceeds - // _budget * BORROW_CEILING_MULTIPLIER. + // Budget-borrowing: hard cap on how far updatePacing() may grow + // (_budget + _borrowed_budget) above _budget alone - _borrowed_budget + // itself is clamped so the resulting ceiling never exceeds + // _budget * BORROW_CEILING_MULTIPLIER. _budget remains a real ceiling in + // the sense that it still bounds how much headroom borrowing can ever + // reach; this only relaxes "never exceeded" into "never exceeded by more + // than a bounded, revocable multiple", which is the whole point of the + // extension (see _borrowed_budget's own comment). static constexpr int BORROW_CEILING_MULTIPLIER = 4; - // Budget-borrowing: fraction of _budget by which _borrowed_budget grows on each pass once - // BORROW_WARMUP_PASSES has been reached - a fraction of the configured budget rather than of the - // current borrowed amount, so growth stays linear (predictable, boundable within a known number - // of passes) rather than compounding. + // Budget-borrowing: fraction of _budget by which _borrowed_budget grows on + // each pass once BORROW_WARMUP_PASSES has been reached - a fraction of the + // configured budget rather than of the current borrowed amount, so growth + // stays linear (predictable, boundable within a known number of passes) + // rather than compounding. static constexpr double BORROW_GROWTH_FRACTION = 0.25; - // Agent-owned BFS thread; startThread()/stopThread() own its lifecycle. + // Agent-owned BFS thread (design doc's Triggering section: "an agent-owned, + // already-attached thread ... calling FollowReferences/IterateThroughHeap + // directly; the safepoint is a side effect of that call, not something the + // profiler builds or schedules"). threadLoop() mirrors J9WallClock's + // pthread lifecycle (J9WallClock, j9/j9WallClock.cpp) rather than BaseWallClock's, + // since J9WallClock's is the simpler of the two shapes actually used for a + // single dedicated thread in this codebase. threadLoop() implements the + // actual scheduling loop (shouldRunPass() below). + // + // start()/stop() themselves still do NOT create/join this thread - + // threadLoop()'s VM::attachThread() call crashes on a null VM::_vm if the + // VM is not yet attached, and referenceChains_ut.cpp calls start() + // directly with no live JVM, so spawning unconditionally from start() + // would crash that gtest binary. startThread()/stopThread() (public API + // above) own the thread's lifecycle instead, and are called from + // Profiler::start()/stop() (profiler.cpp) - the only place in this + // codebase that also calls ReferenceChainTracker::start()/stop() itself, + // and only once the JVM/JVMTI environment is already up. onGCFinish() + // below wakes this thread via pthread_kill(WAKEUP_SIGNAL) whenever it is + // running (i.e. once startThread() has been called) - inert otherwise. pthread_t _thread; - // std::atomic rather than plain volatile bool - volatile alone gives no C++ memory-model - // acquire/release guarantees (it only prevents the compiler from eliding/reordering that one - // variable's own accesses), so a weakly-ordered CPU (e.g. arm64) could let the BFS thread's - // stopThread()- side write (see stopThread()'s own comment) become visible to threadLoop() later - // than intended, missing the shutdown request on one wakeup and sleeping/looping an extra cycle - // before pthread_join() unblocks it. + // std::atomic rather than plain volatile bool - volatile alone gives + // no C++ memory-model acquire/release guarantees (it only prevents the + // compiler from eliding/reordering that one variable's own accesses), so a + // weakly-ordered CPU (e.g. arm64) could let the BFS thread's stopThread()- + // side write (see stopThread()'s own comment) become visible to + // threadLoop() later than intended, missing the shutdown request on one + // wakeup and sleeping/looping an extra cycle before pthread_join() unblocks + // it. Written with memory_order_release from startThread()/stopThread(), + // read with memory_order_acquire from threadLoop() - the same cross-thread + // shape _abort_pass_requested above already uses atomic for. std::atomic _running; - // Cooperative-cancellation flag for an in-flight JVMTI FollowReferences walk: stopThread() sets - // this before pthread_kill()/pthread_join() (that signal alone cannot interrupt a call already - // inside the JVM/JVMTI implementation), and heapReferenceCallback() checks it on every - // invocation, aborting the walk within one callback rather than letting pthread_join() block - // until the walk finishes on its own - see both methods' own comments. + // Cooperative-cancellation flag for an in-flight JVMTI FollowReferences + // walk: stopThread() sets this before pthread_kill()/pthread_join() + // (that signal alone cannot interrupt a call already inside the JVM/JVMTI + // implementation), and heapReferenceCallback() checks it on every + // invocation, aborting the walk within one callback rather than letting + // pthread_join() block until the walk finishes on its own - see both + // methods' own comments. startThread() resets it back to false, since a + // dynamic-attach profiler can cycle through multiple start()/stop() calls + // in one JVM lifetime and a stale abort request would instantly kill the + // next cycle's very first pass. std::atomic: written from + // stopThread()/startThread() on the calling (shutdown) thread and read + // from heapReferenceCallback() on the BFS thread - the same cross-thread + // shape _running above already has, just made explicit via atomic rather + // than a plain volatile bool. std::atomic _abort_pass_requested; - // Wall-clock deadline for the pass currently in flight (OS::nanotime() ticks; 0 = no deadline). + // Wall-clock deadline for the pass currently in flight (OS::nanotime() + // ticks; 0 = no deadline). Set once at the top of runPassManualWalk() from + // _pause_target_ms and shared across that same call's static-field sweep + // and expandFrontier() calls (both read it via heapReferenceCallback()'s + // own periodic check), and collectStaleExpandedEntriesForRotation()'s + // candidate scan (its own periodic check, same amortization pattern - that + // scan is plain C++ under _frontier's shared lock, not a JVMTI/STW call + // itself, but an unbounded scan there would still steal from this same + // pass's wall-clock share before the actual walk even starts). Deliberately + // NOT applied to root/stack-ref + // enumeration (heapRootCallback()) - a live experiment truncating that + // call early on a wall-clock basis measurably reduced total edges admitted + // over a fixed test window versus letting it run to its own (much larger) + // edge budget, because the call's fixed root-walk-and-dispatch cost is paid + // in full regardless of how early it's cut off - see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment for the mechanism that now + // controls how often that call runs instead. Single-threaded: only + // threadLoop() ever calls runPassManualWalk(), so this needs no atomicity. u64 _pass_deadline_ns = 0; - // Last time root/stack-ref enumeration actually ran (OS::nanotime() ticks; 0 before the search's - // first pass). + // Last time root/stack-ref enumeration actually ran (OS::nanotime() ticks; + // 0 before the search's first pass). runPass() compares this against + // ROOT_ENUM_MIN_INTERVAL_NS to decide whether the current pass re-runs + // IterateOverReachableObjects or takes the cheap expandFrontier()-only + // path over the already-persisted frontier. Single-threaded, same as + // _pass_deadline_ns above. u64 _last_root_enum_ns = 0; - // Set true when the most recent root/stack-ref enumeration attempt ended via BUDGET_EXHAUSTED - // (not FRONTIER_CAP_HIT, which stops admitting new frontier entries but leaves the search RUNNING - // - see runPass()'s frontier_cap_hit handling) - runPass() treats this as grounds to retry root - // enumeration on the very next pass regardless of ROOT_ENUM_MIN_INTERVAL_NS, so a - // still-incomplete attempt is not left waiting out the full interval before continuing. + // Set true when the most recent root/stack-ref enumeration attempt ended + // via BUDGET_EXHAUSTED (not FRONTIER_CAP_HIT, which stops admitting new + // frontier entries but leaves the search RUNNING - see runPass()'s + // frontier_cap_hit handling) - runPass() treats this as grounds to retry + // root enumeration + // on the very next pass regardless of ROOT_ENUM_MIN_INTERVAL_NS, so a + // still-incomplete attempt is not left waiting out the full interval + // before continuing. Cleared as soon as an attempt completes without + // truncating. bool _root_enum_truncated_last_time = false; ReferenceChainTracker() @@ -708,7 +2246,7 @@ class ReferenceChainTracker { _hop_cap(0), _budget(0), _first_pass_budget(0), _ttl_ms(0), _pause_target_ms(0), _effective_pause_target_ms(0), _passes_since_last_progress(0), _passes_since_last_candidate_progress(0), _last_candidate_progress_mark(0), - _canary_backoff_mult(1), _canary_pass_ema_ms(0), + _canary_backoff_mult(1), _canary_pass_ema_ns(0), _last_canary_pass_ns(0), _oom_ramp_active(false), _canary_stuck_restart_count(0), _effective_budget(0), _effective_cadence_ns(PASS_CADENCE_NS), @@ -735,95 +2273,311 @@ class ReferenceChainTracker { } void threadLoop(); - // Runs a pass when the GC-finish epoch advanced or the cadence elapsed. + // Combines Open Question 5's two candidate pass-scheduling triggers + // (design doc's Triggering section) rather than picking one: true if the + // GC-finish epoch has advanced since the last pass ("a GC just happened, a + // pass may be worth running soon") or PASS_CADENCE_NS has elapsed since + // the last pass, whichever comes first. Also true before the first pass + // has ever run. A future measurement pass decides whether one of these + // two triggers should be dropped as unnecessary once real cost data + // exists - for now both are implemented, combined, rather than adding an + // unmeasured config knob to switch between them. bool shouldRunPass(u64 now_ns); - // Cheap probe (max=1, not the real poll pollWatchedTargets() makes) into LivenessTracker's - // population-trend table: true if at least one klass shows a positive population slope worth - // chasing. + // Cheap probe (max=1, not the real poll pollWatchedTargets() makes) into + // LivenessTracker's population-trend table: true if at least one klass + // shows a positive population slope worth chasing. Always true when + // LivenessTracker::gcGenerationsEnabled() is off, since there is no + // candidate signal to gate on in that mode - callers fall back to their + // pre-existing behavior in that case. Shared by canAffordNewSearch() + // (restart gate) and threadLoop()'s own steady-state gate below, so a GC + // with no accompanying population growth doesn't trigger either a restart + // or a fresh pass. + // + // Also true, independent of the per-klass check above, while isUrgent() is + // latched and this urgency episode has not yet authorized a search + // (_urgent_search_spent) - see OOM_URGENT_THRESHOLD_S's own comment for why + // the per-klass gate alone is too slow for an aggressive, heap-wide leak, + // and _urgent_search_spent's for why the shortcut is limited to one search + // per episode. This only removes *this* gate: canAffordNewSearch() (the + // actual restart/first-search decision) still checks the pain budget before + // ever calling this method, so a search already cooling down from a recent + // one's own cost can still be deferred even while this returns true. bool hasLeakSignal(); - // Latched, hysteretic view of LivenessTracker::secondsToOOM() crossing OOM_URGENT_THRESHOLD_S - - // see _urgent_latched for the latch/release rules and why the raw comparison flaps. + // Latched, hysteretic view of LivenessTracker::secondsToOOM() crossing + // OOM_URGENT_THRESHOLD_S - see _urgent_latched for the latch/release rules + // and why the raw comparison flaps. When true, runPass() suppresses TTL + // abandonment and threadLoop() tightens cadence + raises the pause + // target so the search completes before the app OOMs. The only SLO is + // the STW pause time, bounded by URGENT_PAUSE_TARGET_MS. const, but + // maintains the latch state (declared mutable) as a side effect, so it + // must be called on every scheduling tick to advance the release counter. bool isUrgent() const; - // Search restart gate (this class's own header comment): true once _safepoint_pain_budget has - // drained back to zero (canStartNow()) *and* hasLeakSignal() above reports at least one leak - // candidate. + // Search restart gate (this class's own header comment): true once + // _safepoint_pain_budget has drained back to zero (canStartNow()) *and* + // hasLeakSignal() above reports at least one leak candidate. Also reused by + // shouldRunPass() to gate the very first search, not just restarts - the + // pain-budget half is always a no-op there (nothing has been spent yet). + // Always true when LivenessTracker::gcGenerationsEnabled() is off, since + // there is no candidate signal to gate on in that mode (see the header + // comment's last paragraph), so a reference-chains-without-generations + // setup is unaffected either way. bool canAffordNewSearch(u64 now_ns); - // Resets every per-search field back to its just-constructed value so the next runPass() call - // takes the "first pass of a search" branch again, exactly like a fresh ReferenceChainTracker - // would. + // Resets every per-search field back to its just-constructed value so the + // next runPass() call takes the "first pass of a search" branch again, + // exactly like a fresh ReferenceChainTracker would. Called by + // shouldRunPass() once a terminal search's tags have already been released + // (runPass() calls releaseSearchTags() itself before returning, so that + // has always already happened by the time this runs) and + // canAffordNewSearch() has approved a restart. Spends the finishing + // search's accumulated cost into _safepoint_pain_budget first, so the *next* + // restart's gate reflects what this one actually cost. Does not touch + // _class_tags/the shared class-tag counter (classTagAllocator.h) - + // classes do not change identity + // across searches, so their resolved names stay valid and do not need + // re-resolving (mirrors _frontier's own stop()/start()-survival + // rationale). frontierTable()'s own resetForRestart() keeps the + // table's allocation but marks every slot unoccupied again, so tags + // restarting from 1 (nextTag()'s only outstanding scheme) do not read back + // stale metadata from the previous search. void restartSearch(); - // Marks every entry still queued in _pending_expand EXPANDED and drains the queue. + // Marks every entry still queued in _pending_expand EXPANDED and drains the + // queue. Called after a *first-pass*, root-seeded FollowReferences call + // that completed without truncation: an uninterrupted walk from the heap + // roots already visits every admitted object's own outgoing edges inline + // (as part of the same call, not a separate one per object - see + // heapReferenceCallback()'s own comment), so nothing is left FRONTIER by + // accident; this just makes that explicit so a later resumed pass has + // nothing pending to expand. void markAllFrontierExpanded(); - // Resolves pending frontier entries via GetObjectsWithTags (dead objects prune for free), then - // expands each with FollowReferences. + // Resumed-pass counterpart to the first pass's root-seeded FollowReferences + // call in runPass(): resolves every not-yet-expanded entry queued in + // _pending_expand via GetObjectsWithTags - design doc Algorithm step 2's + // "resolve currently- + // live tagged frontier objects; objects that fail to resolve are dropped + // (dead - free pruning)" - then calls FollowReferences with the resolved + // object as initial_object to discover its own outgoing edges, exactly as + // the root walk does inline for a first pass. Repeats over newly- + // discovered entries within the same call, stopping the moment + // *edges_admitted reaches `budget` or the frontier table reports capacity + // exhaustion (the same "abort expansion past the cap rather than + // discovering-then-discarding" rule the root-seeded path already follows) + // - leaving the remaining range untouched for a later call to retry. + // + // *frontier_cap_hit distinguishes "budget for this call ran out" (normal; + // the search stays RUNNING, more work remains for the next pass) from + // "the frontier table itself is full" (design doc: "stop admitting new + // entries ... report it" - runPass() treats this as grounds to ABANDON the + // whole search, not just truncate this pass). If GetObjectsWithTags itself + // fails, *truncated is set (there is pending work, just not resolvable + // this call) so runPass() does not mistake that for the search having + // reached natural completion. + // safepoint_ticks: added to (not overwritten - callers may invoke this + // more than once per pass, e.g. ordinary expansion then rotation) with the + // TSC::ticks() duration of just this call's FollowReferences invocation(s) + // - the genuine in-safepoint VM_HeapWalkOperation cost, excluding + // GetObjectsWithTags (not a safepoint call - its own comment above) and + // every other bookkeeping line in this function. See runPass()'s own + // comment for why this needed splitting out from the whole call's + // wall-clock time. void expandFrontier(jvmtiEnv *jvmti, JNIEnv *jni, int hop_cap, int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks); // Static-field counterpart to heapRootCallback()'s GC-root enumeration: - // IterateOverReachableObjects's root/stack-ref callbacks never report a class's static fields - // (there is no jvmtiHeapRootKind for STATIC_FIELD - translateHeapRootKind()'s own comment), so - // without this call an object retained only via `SomeClass.staticField` is never discovered by - // either root enumeration or expandFrontier() (which only descends from already-admitted, - // non-class frontier entries - class objects are never admitted, see heapReferenceCallback()'s - // own comment). + // IterateOverReachableObjects's root/stack-ref callbacks never report a + // class's static fields (there is no jvmtiHeapRootKind for STATIC_FIELD - + // translateHeapRootKind()'s own comment), so without this call an object + // retained only via `SomeClass.staticField` is never discovered by either + // root enumeration or expandFrontier() (which only descends from + // already-admitted, non-class frontier entries - class objects are never + // admitted, see heapReferenceCallback()'s own comment). This drives one + // batched FollowReferences(initial_object=) call - + // mirroring expandFrontier()'s array-holder batching, one FollowReferences + // for every loaded class rather than one per class - so + // heapReferenceCallback()'s existing referrer-is-a-pre-tagged-class + // ("rtag < 0") root-like handling actually gets invoked. An empty + // batch_tags set forces exactly one hop past each class, exactly like + // expandFrontier()'s per-level batching: each admitted static-field + // referent becomes an ordinary frontier entry that a later + // expandFrontier() call expands on its own turn. Best-effort: on any + // failure (no JNIEnv, OOM/local-ref exhaustion building the holder array, + // JVMTI error) this simply skips the sweep for this pass rather than + // treating it as this pass's own truncation - it is discovery on top of + // the manual walk, not part of its budget/frontier-cap accounting. + // safepoint_ticks: same accumulate-not-overwrite contract as + // expandFrontier()'s own parameter above - added to with just this call's + // FollowReferences duration. + // + // Chunked and resumable via _static_field_sweep_cursor + // (STATIC_FIELD_SWEEP_CHUNK_CLASSES classes per call, not every loaded + // class at once): a JVM with tens of thousands of loaded classes cannot + // have its entire static-field graph walked by one FollowReferences call + // within a single pass's 5-50ms safepoint deadline (confirmed on a live + // pod - see doc/temp/ investigation notes - truncated=1 on 275/275 + // observed passes, 0-1 edges admitted out of ~34k classes' worth of + // static fields). Restarting from class 0 every truncated attempt, as a + // single-call sweep must, means whichever classes come after wherever the + // deadline hits are structurally unreachable no matter how many times it + // retries. Chunking instead makes guaranteed forward progress through the + // loaded-class list across passes regardless of any one chunk truncating. + // The holder array is filled in reversed order so HotSpot's LIFO + // FollowReferences descent visits classes in ascending original index + // order; on truncation the cursor resumes at the class that was being + // processed (not chunk_end), so classes after the interruption point are + // reached on the next pass rather than skipped for the rest of the lap. + // Each call also reprioritizes the loaded-class list app-classes-first + // before selecting its chunk (see the .cpp body) so a likely leak source + // is reached within the first several chunks instead of only after every + // JDK/platform class has been swept. *cycle_complete is set when this + // call's chunk reaches + // the end of the loaded-class list (a full lap), which the caller uses + // in place of the old single-call "not truncated" check to decide whether + // to update _last_static_field_class_count. void admitStaticFieldRoots(jvmtiEnv *jvmti, JNIEnv *jni, int hop_cap, int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, bool *cycle_complete, u64 *safepoint_ticks); - // Clears every live JVMTI tag this search still owns; frontier metadata is kept so - // reconstructChain() keeps working. + // Clears the live JVMTI tag (via clearTag(), i.e. SetTag(obj, 0)) for + // every FrontierTable entry this search has not already marked ABANDONED - + // design doc's Termination section: "on abandonment or completion, every + // JVMTI tag this search assigned ... must be cleared before the search's + // state is discarded." Does not discard the FrontierTable's own records + // (referrer_klass/parent_tag/depth survive, so reconstructChain() keeps + // working from memory after the search ends) - only the underlying + // object's live JVMTI tag is released, via the same batch + // resolve-then-clear sequence GetObjectsWithTags makes possible for + // expandFrontier()'s resolve-or-drop path. + // + // Returns true if every live tag scanned this call was successfully + // resolved-and-cleared (or there were none to begin with), false if + // GetObjectsWithTags() itself failed - in which case NO entry is marked + // ABANDONED (unlike a resolve failure for an individual tag, which means + // the object is already dead and safe to treat as released): a batch + // GetObjectsWithTags() failure tells us nothing about which, if any, + // objects in the batch are still live, so marking them ABANDONED here + // would let restartSearch() reset _next_tag/the frontier table while a + // still-live object could still be holding this search's JVMTI tag, + // corrupting the next search's tag-uniqueness invariant. Callers must not + // allow a restart until this returns true; calling it again later safely + // retries only the tags still not marked ABANDONED from a prior failed + // call. bool releaseSearchTags(jvmtiEnv *jvmti, JNIEnv *jni); - // Feeds the measured pass duration into _pause_pid and rescales the effective budget and - // cadence; _budget remains the hard ceiling. + // Pause-time pacing controller: feeds `pass_wall_ns` - the wall-clock duration of the FollowReferences/ + // GetObjectsWithTags call runPass() just made (the safepoint-triggering + // call itself, per the design doc's Triggering section; no new + // instrumentation needed, since this class is already the thread blocked + // inside it) - into `_pause_pid`, and scales `_effective_budget`/ + // `_effective_cadence_ns` from its output. Folds Open Questions 2 and 5 + // into the one controller call rather than two + // separately tuned mechanisms. + // + // `_pause_pid.compute()`'s sign convention (pidController.cpp): a positive + // signal means the measured value came in *under* the controller's + // target - here, the last pass finished comfortably inside + // `_pause_target_ms`, so there is headroom to admit a larger budget next + // time. This is the opposite of ObjectSampler/MallocTracer/ + // NativeSocketSampler's own usage (ObjectSampler::updateConfiguration(), + // MallocTracer::_pid, rateLimiter.h), which *subtract* the signal + // from their interval because their controlled variable (a sampling + // interval) is inversely related to their target rate; `_effective_budget` + // is directly related to pass duration (more budget -> longer pass), so + // the signal is *added* here instead. + // + // The result is clamped to [floor, _budget] - `_budget` (the config value, + // Arguments::_reference_chains_budget) becomes this controller's ceiling + // rather than a fixed per-pass value, clamped to never above the + // frontier/hop caps: the hop cap and frontier cap stay untouched, + // fixed correctness bounds exactly as before (design doc: "not + // controller-tuned"). Whatever part of the signal the clamp could not + // absorb (`overflow` below) drives `_effective_cadence_ns` instead - the + // cadence decision folded into the same controller output rather than a + // second mechanism: a search still running long even at the + // minimum budget backs off the fallback cadence instead of trying to + // shrink the budget further (avoiding a degenerate near-zero budget just + // to hit an aggressive cadence); a search with + // spare headroom even at the maximum (config) budget relaxes the cadence + // toward MIN_EFFECTIVE_CADENCE_NS instead, letting the GC-finish-epoch + // trigger (shouldRunPass(), already unconditional on cadence) make + // progress as often as it fires. void updatePacing(u64 pass_wall_ticks); - // Root/stack-ref enumeration passes never reach updatePacing() (runPass()'s own comment: their - // fixed dispatch cost would wrongly throttle _effective_budget for every unrelated later pass), - // but a slow one still spends real pause-time-SLO budget the borrow ceiling promised was safe to - // hand out. + // Root/stack-ref enumeration passes never reach updatePacing() (runPass()'s + // own comment: their fixed dispatch cost would wrongly throttle + // _effective_budget for every unrelated later pass), but a slow one still + // spends real pause-time-SLO budget the borrow ceiling promised was safe to + // hand out. _borrowed_budget's own comment requires it be revoked the + // instant ANY pass is not comfortably under target, so this call - made in + // updatePacing()'s place for a root-enum pass - only ever revokes, never + // grows the streak/borrow: the warmup counter is calibrated against + // expandFrontier()'s per-node cost, not this call's unrelated fixed cost. void maybeRevokeBorrowForRootEnumPass(u64 pass_wall_ticks); - // Tags every not-yet-tagged loaded class (GetLoadedClasses()) with a fresh nextClassTag() and - // resolves its name into _class_tags, via the same GetClassSignature + normalizeClassSignature + - // Profiler::lookupClass sequence ObjectSampler::recordAllocation() already uses - // (objectSampler.cpp:76-90) - reusing that normalization helper rather than re-deriving it. + // Tags every not-yet-tagged loaded class (GetLoadedClasses()) with a + // fresh nextClassTag() and resolves its name into _class_tags, via the + // same GetClassSignature + normalizeClassSignature + Profiler::lookupClass + // sequence ObjectSampler::recordAllocation() already uses + // (objectSampler.cpp:76-90) - reusing that normalization helper rather + // than re-deriving it. Run once at the start of every runPass() (before + // FollowReferences) rather than lazily during the walk, because + // GetClassSignature/JNI calls are not allowed from inside + // heapReferenceCallback() (see the file header comment) - by pre-tagging, + // every class_tag the callback sees is already resolvable with no further + // JVMTI/JNI calls of its own. Already-tagged classes (from a previous + // pass) are skipped, not re-resolved. void resolveLoadedClasses(jvmtiEnv *jvmti, JNIEnv *jni); - // jvmtiHeapReferenceCallback for runPass()'s FollowReferences call (see runPass() below for the - // full walk). + // jvmtiHeapReferenceCallback for runPass()'s FollowReferences call (see + // runPass() below for the full walk). `user_data` is a PassContext* + // (referenceChains.cpp, private to the .cpp - the type never needs to be + // visible here since only runPass() constructs one). static jint JNICALL heapReferenceCallback( jvmtiHeapReferenceKind reference_kind, const jvmtiHeapReferenceInfo *reference_info, jlong class_tag, jlong referrer_class_tag, jlong size, jlong *tag_ptr, jlong *referrer_tag_ptr, jint length, void *user_data); - // Outcome of admitObject() below - lets each of its two call sites (heapReferenceCallback() above - // and the IterateOverReachableObjects root/ stack-ref callbacks, both in referenceChains.cpp) - // translate the same admission decision into its own callback-shape-appropriate return + // Outcome of admitObject() below - lets each of its two call sites + // (heapReferenceCallback() above and the IterateOverReachableObjects root/ + // stack-ref callbacks, both in referenceChains.cpp) translate the same + // admission decision into its own callback-shape-appropriate return // value/truncation flag, instead of duplicating the decision twice. enum class AdmitResult { ALREADY_ADMITTED, // *tag_ptr != 0: nothing to do, not a truncation HOP_CAP, // depth >= hop_cap: not admitted, not a truncation BUDGET_EXHAUSTED, // edges_admitted >= budget: this pass's cap FRONTIER_CAP_HIT, // FrontierTable::insert() itself is full: stops - // admitting new entries but does not itself abandon the search (see - // runPass()'s frontier_cap_hit handling - the no-progress detector abandons - // only if the frontier then stops growing) + // admitting new entries but does not itself abandon + // the search (see runPass()'s frontier_cap_hit + // handling - the no-progress detector abandons only + // if the frontier then stops growing) ADMITTED, }; - // First-discovery admission core: factored out of heapReferenceCallback()'s inline admission - // branch so the manual-walk driver's root/stack-ref callbacks stay in sync with FollowReferences' - // own admission by construction, not by copy-paste. + // First-discovery admission core: + // factored out of heapReferenceCallback()'s inline admission branch so the + // manual-walk driver's root/stack-ref callbacks stay in sync with + // FollowReferences' own admission by construction, not by copy-paste. + // `*tag_ptr` is the in/out tag slot each call site already has as a real + // JVMTI out-parameter, and `*edges_admitted` + // is likewise each call site's own running counter for this call. On + // AdmitResult::ADMITTED, `*tag_ptr` is filled with the freshly assigned tag + // and the tag is queued onto _pending_expand (or, when `priority` is true, + // onto _priority_expand instead - see that field's own comment) exactly as + // heapReferenceCallback() already did inline. + // `class_tag` is the raw JVMTI class tag of the object being admitted - + // both real call sites (heapReferenceCallback()/heapRootCallback()) have + // this in hand already as their own JVMTI callback parameter, so no new + // JVMTI call is needed to supply it. Stored into the new entry's + // FrontierEntry::class_tag (see that field's own comment) and forwarded + // to trackLeakAccumulation() below. AdmitResult admitObject(FrontierTable *frontier, int hop_cap, int budget, int *edges_admitted, jlong *tag_ptr, jlong parent_tag, u32 referrer_klass, u32 depth, @@ -832,112 +2586,313 @@ class ReferenceChainTracker { jint edge_field_index = -1, u8 edge_kind = 0, jlong edge_referrer_class_tag = 0); - // Called by admitObject() on every successful ADMITTED result (root or non-root, ordinary or - // priority) - the single shared admission path, so this needs no duplicate call site at - // heapReferenceCallback()/ heapRootCallback()/stackRefCallback(). + // Called by admitObject() on every successful ADMITTED result (root or + // non-root, ordinary or priority) - the single shared admission path, so + // this needs no duplicate call site at heapReferenceCallback()/ + // heapRootCallback()/stackRefCallback(). O(1) in the common case + // (_watched_leak_klass_count == 0, before any leak signal has fired - + // just one integer compare) and O(MAX_WATCHED_LEAK_KLASSES) plus one + // frontier lookup when something is being watched - see + // _leak_signature_totals/_leak_parent_fanout's own comments for what this + // actually records and collectLeakAccumulationCandidatesForRotation()'s + // own comment for the full design this feeds. `class_tag` is the newly- + // admitted object's own raw JVMTI class tag (FrontierEntry::class_tag), + // NOT referrer_klass (a classMap dictionary id) - see class_tag's own + // comment for why matching against _watched_leak_klass_ids requires the + // stable tag, not the compactable dictionary id. void trackLeakAccumulation(FrontierTable *frontier, jlong class_tag, jlong parent_tag, jlong tag); - // Record a discovered instance for a watched candidate class: store its frontier tag in the - // class's discovery slots so pollWatchedTargets() can build its chain event. + // Record a discovered instance for a watched candidate class: store its + // frontier tag in the class's discovery slots so pollWatchedTargets() can + // build its chain event. Shared by the ordinary auto-mark path + // (heapReferenceCallback on admission), the leak-tag interception path, + // and correlateAdmittedLeakTag() below - the latter two pass + // leak_correlated=true, which lets them EVICT a slot held by an + // uncorrelated (noise) instance when all slots are full. Slots are bounded + // by MAX_DISCOVERED_INSTANCES_PER_CLASS and noise instances can fill them + // before the leak-tagged ones are ever reached - without preferential + // eviction the tracked leak instances would be silently discarded + // (observed live: 8 depth-1 jni_local/stack_local noise instances + // permanently occupying all slots of the watched [B class). + // Single-thread: only called from the pass thread (heapReferenceCallback) + // and the tracker's own thread (correlateAdmittedLeakTag via + // tagLeakInstances from pollWatchedTargets) - never concurrently. + // Builds and caches chain events for every discovered instance recorded + // against a slot holding klass_id - slot-driven so instances recorded while + // a candidate still qualified are not stranded when it stops qualifying + // (see the definition's own comment in referenceChains.cpp for the observed + // live failure this closes and the two call sites). void buildDiscoveredInstanceChains(jvmtiEnv *jvmti, JNIEnv *jni, u32 klass_id, u64 current_search_ns); void recordDiscoveredInstance(u32 klass_id, jlong frontier_tag, bool leak_correlated); - // Correlate a leak tag with an instance the BFS admitted BEFORE tagLeakInstances() tagged it (its - // JVMTI tag is a frontier tag, its frontier entry has leak_tag == 0). - - // One-time retroactive catch-up for a klass_id the moment it FIRST enters _watched_leak_klass_ids - // (pollWatchedTargets() calls this only for the newly-added ids in each refresh, never for ones - // already being watched). + // Correlate a leak tag with an instance the BFS admitted BEFORE + // tagLeakInstances() tagged it (its JVMTI tag is a frontier tag, its + // frontier entry has leak_tag == 0). Sets the entry's leak_tag so chain + // events emit targetTag = the leak tag (the HeapLiveObject correlation + // key), and records the instance as discovered. Returns false if the tag + // resolves to no live frontier entry (caller should treat the object as + // un-tagged). Idempotent: an entry that already carries a leak tag just + // returns true. Also handles post-restart re-admission: the new entry for + // a re-admitted instance gets the SAME leak tag the pool already holds + // for it (LivenessTracker's record survives the search restart). + // Public: LivenessTracker::tagLeakInstances() (livenessTracker.cpp) + // calls it - see the public section below for the declaration. + + // One-time retroactive catch-up for a klass_id the moment it FIRST enters + // _watched_leak_klass_ids (pollWatchedTargets() calls this only for the + // newly-added ids in each refresh, never for ones already being watched). + // trackLeakAccumulation() above only fires on NEW admissions + // (admitObject()'s ADMITTED result) - it cannot see objects that were + // already admitted before this klass_id started being watched, which for + // a klass that has been growing for a while (the exact case this + // mechanism targets) can be nearly all of them, found the hard way: the + // container that actually needs re-expansion typically already got fully + // admitted in an early pass, long before selectLeakCandidates()'s own + // hysteresis gate ever authorized watching it, leaving + // _leak_signature_totals/_leak_parent_fanout permanently empty with no + // way to ever get their first data point. This scans the WHOLE frontier + // table once (bounded by table size, same cost class as + // collectStaleExpandedEntriesForRotation()'s existing per-pass scan, but + // this one runs only on the rare newly-watched-klass event - at most + // MAX_WATCHED_LEAK_KLASSES times per search, not every pass) for + // already-EXPANDED entries whose (u32) class_tag matches klass_id (both + // sides of the comparison are truncated the same way - see + // _watched_leak_klass_ids' own comment for why a full jlong is not + // needed), and feeds each one through the same aggregation logic + // trackLeakAccumulation() uses for new admissions, applied retroactively - + // so ongoing incremental updates compose cleanly on top of this baseline + // without double-counting. void seedLeakAccumulationForNewlyWatchedKlass(u32 klass_id); - // Upgrades root_kind when a rediscovered heap root is more durable; testable without a JVMTI mock. + // Durability tie-break (design doc's "Fix for root-attribution staleness" + // point 1) for an object rediscovered as a heap root by + // heapRootCallback()/stackRefCallback() while already admitted (same pass + // or a previous one) - factored out of those callbacks, rather than + // inlined, so it is unit-testable without a PassContext/JVMTI mock (both + // callbacks' user_data type is private to referenceChains.cpp). Only ever + // overwrites root_kind - never parent_tag - and only for an entry that is + // already root-attached (entry.parent_tag == 0); this is the option (a) + // resolution of the parent_tag==0/root_kind invariant conflict: a + // re-expansion-driven, non-root rediscovery of an edge to + // some already-tracked, non-root-attached object must never reach this + // method at all (re-expansion child admission always passes + // root_kind=0, which loses every tie-break, so it structurally cannot + // trigger an upgrade even if it were mistakenly routed here). Returns true + // if an upgrade was applied, false otherwise (already at least as durable, + // not root-attached, or not found) - purely informational for callers/ + // tests, not required for correctness. bool maybeUpgradeRootAttachedRootKind(FrontierTable *frontier, jlong tag, u8 new_root_kind); - // True if `tag` is already sitting in _priority_expand - either queued earlier this same pass by - // the other rotation collector, or left over from a prior pass's truncated batch - // (expandFrontier() leaves those at the front of the queue for a later retry rather than popping - // them). + // True if `tag` is already sitting in _priority_expand - either queued + // earlier this same pass by the other rotation collector, or left over + // from a prior pass's truncated batch (expandFrontier() leaves those at + // the front of the queue for a later retry rather than popping them). + // Shared by both rotation collectors below so neither can push a tag + // that's already pending re-expansion. O(1) via _priority_expand_set - + // the rotation collectors run this check for EVERY FrontierTable slot + // they visit (~199k EXPANDED entries on a large heap), so the original + // linear scan over the deque was ~200M comparisons per rotation pass at + // the PRIORITY_EXPAND_CAP (observed prominently in profiles); the + // "sub-millisecond" claim its original comment made only held for the + // per-SELECTION calls it was written for, not the per-slot visits the + // collectors actually make. bool isQueuedForRotation(jlong tag) const { return _priority_expand_set.contains(tag); } - // Re-queues transient-rooted expanded entries so a durable root can supersede a stale - // attribution. + // Bounded rotating re-expansion (design doc's closing section): + // each manual-walk pass, feed up to `max_count` already-EXPANDED, + // root-attached entries whose root_kind is still transient + // (isTransientRootKind()) back into _priority_expand so + // expandFrontier() re-walks their fields - giving a stale root_kind + // another chance to be superseded by a durable root discovered elsewhere + // in the interim, via the same admitObject()/tie-break machinery every + // other admission uses. Scans FrontierTable slots in tag order starting + // from _root_kind_rotation_cursor, wrapping at size(), so repeated calls + // sweep the whole table over time instead of only ever revisiting the + // first `max_count` transient entries. Pure table scan/queue push - no + // JVMTI call of its own - so it is unit-testable directly. + // Returns the tags selected (also already pushed onto _priority_expand). std::vector collectStaleRootKindEntriesForRotation(int max_count); - // Bounded rotating re-expansion for stale mutable fields: expandFrontier() observes an object's - // outgoing references exactly once (on the FollowReferences call that marks it EXPANDED) and - // never revisits it, so a field that is later reassigned to point at a different object - e.g. - // HashMap.table on resize - has its new value permanently unobserved once the map itself is - // EXPANDED; the old table array's own frontier entry eventually resolves to a dead object via - // GetObjectsWithTags and gets silently cleared with zero children, orphaning everything only - // reachable through the *current* table. + // Bounded rotating re-expansion for stale mutable fields: + // expandFrontier() observes an object's outgoing references exactly once + // (on the FollowReferences call that marks it EXPANDED) and never + // revisits it, so a field that is later reassigned to point at a + // different object - e.g. HashMap.table on resize - has its new value + // permanently unobserved once the map itself is EXPANDED; the old table + // array's own frontier entry eventually resolves to a dead object via + // GetObjectsWithTags and gets silently cleared with zero children, + // orphaning everything only reachable through the *current* table. + // Feeds up to `max_count` already-EXPANDED entries (any parent_tag/ + // root_kind - unlike collectStaleRootKindEntriesForRotation() above, which + // is scoped to root-attached transient entries for a different reason) + // back into _priority_expand so expandFrontier() re-runs FollowReferences + // on them and observes their current field values. Already-admitted + // children are ALREADY_ADMITTED no-ops (admitObject()'s own idempotency); + // only a genuinely new edge (i.e. a mutated field) is admitted. Scans + // FrontierTable slots in tag order starting from + // _stale_expanded_rotation_cursor, wrapping at size() - same rationale as + // collectStaleRootKindEntriesForRotation() above: a fixed always-from-1 + // scan lets a large, permanently-EXPANDED low-tag population (long-lived + // infrastructure objects) monopolize every pass's cap forever, starving + // any higher-tag entry (e.g. a static field's collection, admitted only + // once its class loads) of ever being re-queued. std::vector collectStaleExpandedEntriesForRotation(int max_count); - // Bounded rotating re-expansion targeting the accumulation point of a klass LivenessTracker has - // flagged as growing (LivenessTracker:: topKlassesByGenerationCount(), _watched_leak_klass_ids) - - // the design's actual targeted tier. + // Bounded rotating re-expansion targeting the accumulation point of a + // klass LivenessTracker has flagged as growing (LivenessTracker:: + // topKlassesByGenerationCount(), _watched_leak_klass_ids) - the design's + // actual targeted tier. See its own definition comment (referenceChains.cpp) + // for the full two-tier design (class-level growth ranking, then per- + // parent fanout ranking within the winner) and why depth/root-durability/ + // class-shape heuristics alone were measured and found insufficient. + // Unlike the other two rotation collectors, has no wrapping cursor - it + // always selects the current best candidate(s), which is the desired + // behavior here (re-selecting a still-growing parent every pass), not + // something a fairness-across-passes guarantee needs to correct for. std::vector collectLeakAccumulationCandidatesForRotation( int max_count); // CANDIDATE-SCOPED REACH: bounded descend walk from an anchor object. + // FollowReferences(initial_object=anchor) with batch_tags == nullptr so + // descent is gated only by hop_cap + the pass deadline, reusing + // heapReferenceCallback() unchanged - leak-tag interception, canary + // pruning, auto-mark and improveChain all work as-is, and every admitted + // child chains back to the anchor's own frontier entry, so an interception + // during the walk yields the complete root->...->chunk chain in ONE + // bounded STW call. The walk's ctx.hop_cap is lowered to + // min(_hop_cap, anchor_depth + DESCENT_HOPS), bounding admission to a few + // hops below the anchor. Motivation (pod rounds 5-6, + // ev-leaktag-onpod-round5/6): breadth-first FIFO expansion over a rising + // heap can never drain (_pending_expand net-growing, per-lap sweep + // re-admissions replenishing it), so the tagged leak instances sit under + // holders the crawl reaches only after hours - candidate-scoped walks + // make the reach independent of the backlog. The walk prunes descent + // (and admission) into a small fixed set of fat-metadata classes + // (java.lang.ClassLoader, java.lang.ThreadGroup, + // java.security.ProtectionDomain - exact class-tag match) resolved fresh + // per call: without this, a Thread anchor's contextClassLoader edge would + // descend into every loaded class and its statics, sweep-scale cost per + // thread per pass. When `anchor_descend_class_tag` is non-zero, the + // ANCHOR's own outgoing edges are additionally gated to descend only + // into referees of that exact class (used by walkCandidateThreadLocals() + // to descend only into ThreadLocal$ThreadLocalMap - the value type of + // BOTH of Thread's threadLocals and inheritableThreadLocals fields), + // letting the walk skip the Thread's other (transient, metadata-heavy) + // instance fields entirely. Deliberately a class-tag comparison, NOT + // jvmtiHeapReferenceInfoField.index matching: that index is a jint field + // ordinal whose correspondence to GetClassFields() order the design has + // no need to depend on. void descendFromAnchor(jvmtiEnv *jvmti, JNIEnv *jni, jobject anchor, jlong anchor_tag, u32 anchor_depth, jlong anchor_descend_class_tag, int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks); - // Prong 1 of the candidate-scoped reach design (thread-retained taxonomy: ThreadLocal-held caches - // and thread-owned collections): per pass, walk up to THREAD_WALK_MAX_ANCHORS of the current - // candidates' qualifying tids' live Thread objects (registerThreadObject()'s map above) with - // descendFromAnchor() (anchor-gated to ThreadLocalMap, see above). + // Prong 1 of the candidate-scoped reach design (thread-retained taxonomy: + // ThreadLocal-held caches and thread-owned collections): per pass, walk + // up to THREAD_WALK_MAX_ANCHORS of the current candidates' qualifying + // tids' live Thread objects (registerThreadObject()'s map above) with + // descendFromAnchor() (anchor-gated to ThreadLocalMap, see above). A + // thread-local accumulation's holder chain + // is entirely inside the Thread object's own ThreadLocalMap subgraph, so + // the walk reaches the tagged instances regardless of the ordinary + // BFS backlog - which the whole-heap crawl demonstrably never does in a + // rising heap. Anchor admission is idempotent across passes (GetTag + + // FrontierTable lookup, reuse-or-retag), so re-walking a thread per pass + // is cheap and ALREADY_ADMITTED-safe. void walkCandidateThreadLocals(jvmtiEnv *jvmti, JNIEnv *jni, int budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks); - // Selects root-attached durable entries with a tiered cursor: leak-tagged, fresh, container-shaped, - // then everything else fairly. + // Prong 2 (durable-root-retained taxonomy): select up to `max_count` root- + // attached entries held by a DURABLE root kind (parent_tag == 0, + // root_kind STATIC_FIELD or JNI_GLOBAL, FRONTIER or EXPANDED) with a + // TIERED cursor (leak-tagged first, then FRESH anchors - not yet given + // their one first-look walk priority, the round-15 queue lane - then + // container-shaped anchors, then everything else cursor-fairly — see + // collectStaticFieldAnchorsForRotation()'s own comment) - and + // descend-walk each via walkStaticFieldAnchors(). A static Map/List's + // leaked chunks sit 3-4 hops below its root-attached holder, deeper + // than the one-hop Tier-2 rotation can reach from an un-expanded + // FRONTIER holder; a descend walk covers the holder's whole internal + // structure in one bounded call. JNI_GLOBAL joined the filter after pod + // round 7 (interception zero with statics fully walked): a leaking + // holder retained through a JNI global is the same taxonomy shape and + // the same walk covers it - covering it in the SAME rebuild avoids a + // second deploy cycle if the pod's holder turns out to be one. std::vector collectStaticFieldAnchorsForRotation(int max_count); - // At-risk anchor push: deduped, capped, and quota-limited per klass. + // The B' at-risk push itself (both call sites below): dedupe via the + // set, cap-drop when the FIFO is full, per-class quota drop at + // STATIC_ANCHOR_ATRISK_PER_KLASS_CAP (round 16 - see + // _static_anchor_fifo_klass_counts), count admitted pushes. No return + // value - a dropped push is silently retried by the feed's next event + // (the next static edge onto the entry, or the next demotion). Engine + // thread only. void pushAtRiskStaticAnchor(jlong tag, u32 klass_id); - // Add `tag` to _static_anchor_index if its root_kind is a durable anchor-tier kind (STATIC_FIELD - // or JNI_GLOBAL). + // Add `tag` to _static_anchor_index if its root_kind is a durable + // anchor-tier kind (STATIC_FIELD or JNI_GLOBAL). Called at first + // admission and at root-kind upgrade. `own_class_tag` is the class tag + // of the anchor OBJECT itself (heapReferenceCallback's class_tag param / + // FrontierEntry::class_tag) - kept in the parallel array so selection can + // tier by class shape without JNI. Idempotent (dedup via a linear + // scan of the small vector — the anchor population is O(hundreds), + // well under the 256-element linear-scan cutoff). Engine thread only. void addToStaticAnchorIndex(jlong tag, jlong own_class_tag, u8 root_kind); - // True iff `klass` implements java/util/Collection or java/util/Map, directly or transitively - // (superclass chain + interfaces of every visited class, depth-bounded, visited set to survive - // interface diamonds). + // True iff `klass` implements java/util/Collection or java/util/Map, + // directly or transitively (superclass chain + interfaces of every + // visited class, depth-bounded, visited set to survive interface + // diamonds). The two interface class tags are resolved once and cached + // (_collection_iface_class_tag/_map_iface_class_tag). Caller owns local- + // ref hygiene for the jclasses this walks. Engine thread only. bool classImplementsContainerOrMap(jvmtiEnv *jvmti, JNIEnv *jni, jclass klass); - // Resolve _collection_iface_class_tag/_map_iface_class_tag once; returns false if the interfaces - // cannot be resolved yet (leaves them at -1 so the next call retries). + // Resolve _collection_iface_class_tag/_map_iface_class_tag once; returns + // false if the interfaces cannot be resolved yet (leaves them at -1 so + // the next call retries). Engine thread only. bool resolveContainerInterfaceTags(jvmtiEnv *jvmti, JNIEnv *jni); - // Lazy shape reconciliation for the anchor index: scans _static_anchor_own_class_tags for class - // tags not yet in _class_shape_cache, resolves up to ANCHOR_SHAPE_RECONCILE_BUDGET of them per - // pass via one GetObjectsWithTags call (class objects are tagged with their class tags) and - // classifies each. + // Lazy shape reconciliation for the anchor index: scans + // _static_anchor_own_class_tags for class tags not yet in + // _class_shape_cache, resolves up to ANCHOR_SHAPE_RECONCILE_BUDGET of + // them per pass via one GetObjectsWithTags call (class objects are + // tagged with their class tags) and classifies each. Selection treats + // unclassified anchors as the lowest tier, so classification lag only + // delays a container's promotion - it never drops coverage. Runs on the + // engine thread with JNI available, outside any frontier lock and + // outside heap callbacks. TEMP: also emits the per-pass cohort + // histogram (round-14 measurement: container cohort size vs the ~4k + // per-search walk coverage) - remove once the arithmetic is verified + // on-pod. void reconcileAnchorClassShapes(jvmtiEnv *jvmti, JNIEnv *jni); - // Pops up to max_count AtRiskAnchor entries off _static_anchor_fifo's front into `out` - // (appending), decrementing each popped entry's class occupancy in - // _static_anchor_fifo_klass_counts (erased at zero, so the map tracks the FIFO's live contents), - // and re-derives the set from the deque's remaining contents (PriorityExpandSet's tombstone-free - // rebuildFrom contract). + // Pops up to max_count AtRiskAnchor entries off _static_anchor_fifo's + // front into `out` (appending), decrementing each popped entry's class + // occupancy in _static_anchor_fifo_klass_counts (erased at zero, so the + // map tracks the FIFO's live contents), and re-derives the set from the + // deque's remaining contents (PriorityExpandSet's tombstone-free + // rebuildFrom contract). Returns the drained count. Engine thread only. int drainStaticAnchorFifo(int max_count, std::vector &out); - // Pushes `entries` back to _static_anchor_fifo's FRONT in reverse order (preserving FIFO order), - // re-incrementing each entry's class occupancy, and rebuilds the set - the truncated-walk requeue - // path. + // Pushes `entries` back to _static_anchor_fifo's FRONT in reverse order + // (preserving FIFO order), re-incrementing each entry's class occupancy, + // and rebuilds the set - the truncated-walk requeue path. Caller passes + // ONLY entries it drained from the FIFO this pass (never collector- + // sourced ones: those keep their own cursor retention). Engine thread + // only. void requeueStaticAnchorFifoFront(const std::vector &entries); - // When non-null, receives the tags of RESOLVED-but-unwalked anchors at the truncation break point - // - GetObjectsWithTags may return fewer anchors than requested (dead tags drop out) in its own - // order, so the caller cannot recover the un-walked set from a consumed index; the walk hands the - // exact tags back instead. + // When non-null, receives the tags of RESOLVED-but-unwalked anchors at + // the truncation break point - GetObjectsWithTags may return fewer + // anchors than requested (dead tags drop out) in its own order, so the + // caller cannot recover the un-walked set from a consumed index; the + // walk hands the exact tags back instead. Dead (unresolved) anchors are + // omitted: they must not be requeued anywhere. void walkStaticFieldAnchors(jvmtiEnv *jvmti, JNIEnv *jni, const std::vector &anchor_tags, int budget, int *edges_admitted, bool *truncated, @@ -945,7 +2900,13 @@ class ReferenceChainTracker { std::vector *unwalked = nullptr); // jvmtiHeapRootCallback/jvmtiStackReferenceCallback for runPassManualWalk()'s - // IterateOverReachableObjects call (referenceChains.cpp). + // IterateOverReachableObjects call (referenceChains.cpp). `user_data` is a + // PassContext* (the same private-to-the-.cpp type heapReferenceCallback() + // already uses above) - both callbacks only ever admit a root-attached + // entry (parent_tag=0, depth=0), translating the JVMTI-owned + // jvmtiHeapRootKind into FrontierEntry::root_kind's jvmtiHeapReferenceKind + // numbering first (see referenceChains.cpp's translateHeapRootKind() for + // why this translation is required, not optional). static jvmtiIterationControl JNICALL heapRootCallback(jvmtiHeapRootKind root_kind, jlong class_tag, jlong size, jlong *tag_ptr, void *user_data); @@ -954,28 +2915,72 @@ class ReferenceChainTracker { jlong *tag_ptr, jlong thread_tag, jint depth, jmethodID method, jint slot, void *user_data); - // Manual-walk pass driver: when `run_root_enum` is true, seeds/refreshes root-attached frontier - // entries via IterateOverReachableObjects (heapRootCallback()/stackRefCallback() above) using - // `root_enum_budget`; then, regardless of `run_root_enum`, drains _pending_expand via - // admitStaticFieldRoots()/expandFrontier() up to `expand_budget`. + // Manual-walk pass driver: when + // `run_root_enum` is true, seeds/refreshes root-attached frontier entries + // via IterateOverReachableObjects (heapRootCallback()/stackRefCallback() + // above) using `root_enum_budget`; then, regardless of `run_root_enum`, + // drains _pending_expand via admitStaticFieldRoots()/expandFrontier() up to + // `expand_budget`. The two budgets are independent, not shared - see + // ROOT_ENUM_MIN_INTERVAL_NS's own comment for why root enumeration gets its + // own, much larger, infrequent allowance instead of competing with the + // small steady-state budget every tick's expansion uses. runPass() decides + // `run_root_enum`/`root_enum_budget`; when false, this call takes the + // cheap expandFrontier()-only path over whatever the last enumeration + // already admitted into the frontier. + // *safepoint_ticks is zeroed here, then accumulated (via + // IterateOverReachableObjects's own timing below plus + // admitStaticFieldRoots()/expandFrontier()'s additive parameters) with just + // the genuine in-safepoint JVMTI call cost this pass incurred - runPass() + // uses it (not this whole call's wall-clock time) as the pacing signal. void runPassManualWalk(jvmtiEnv *jvmti, JNIEnv *jni, bool run_root_enum, int root_enum_budget, int expand_budget, int *edges_admitted, bool *truncated, bool *frontier_cap_hit, u64 *safepoint_ticks); - // Inserts (or refreshes) klass_id's resolved chain in _resolved_chains, recording the - // source_tag/source_search_ns it was reconstructed from so a later poll can tell a stale entry - // from a current one. + // Inserts (or refreshes) klass_id's resolved chain in _resolved_chains, + // recording the source_tag/source_search_ns it was reconstructed from so a + // later poll can tell a stale entry from a current one. Drops (counting via + // REFERENCE_CHAIN_EVENTS_DROPPED) rather than evicting when a brand-new + // klass_id arrives with the cache already at MAX_RESOLVED_CHAINS - see that + // constant's own comment and this method's definition (referenceChains.cpp). + // Returns false when the chain was dropped (cache full) so the caller can + // skip coverage accounting for a chain that will never be emitted. bool cacheResolvedChain(jlong source_tag, ReferenceChainEvent &&event, - jlong source_tag_val, u64 source_search_ns); - - // Remove a cached chain so pollWatchedTargets rebuilds it on the next poll. + u64 source_search_ns); + + // Remove a cached chain so pollWatchedTargets rebuilds it on the next + // poll. Called when improveChain updates a frontier entry with a + // deeper path — the cached chain (built from the old shallow entry) + // must be discarded so the deeper chain is emitted instead. + // NOTE: must not be called from inside the heapReferenceCallback() + // FollowReferences pause - it takes _resolved_chains_lock, which + // drainPendingChainEvents() holds across a full cache copy on the JFR + // dump thread, stalling the walk. Use deferResolvedChainInvalidation() + // from callback context and drain on the BFS thread after the pass. void invalidateResolvedChain(jlong source_tag); - // Snapshots the just-abandoned search into _pending_abandoned_events - called from runPass() - // (referenceChains.cpp) immediately after it writes SearchState::ABANDONED, while - // buildAbandonedEvent()'s source fields are still valid (see _pending_abandoned_events' own - // comment for why this cannot be deferred to dump()-time). + // Callback-safe variant: records the tag for later invalidation instead + // of taking _resolved_chains_lock inside the FollowReferences STW pause. + // The pending list is drained by drainPendingChainInvalidations() on the + // BFS thread after the pass (and best-effort at pollWatchedTargets() + // entry). Erasure timing only affects WHEN the stale cached chain stops + // being re-emitted (next drain vs same pass), never correctness. + void deferResolvedChainInvalidation(jlong source_tag); + + // Applies the deferred invalidations recorded by + // deferResolvedChainInvalidation(). BFS-thread only (the list is written + // from the walk callback and the JNI test seam, both serialized by the + // engine lock / seam contract, but drained under its own lock anyway so + // the drain can also run from pollWatchedTargets()). + void drainPendingChainInvalidations(); + + // Snapshots the just-abandoned search into _pending_abandoned_events - + // called from runPass() (referenceChains.cpp) immediately after it writes + // SearchState::ABANDONED, while buildAbandonedEvent()'s source fields are + // still valid (see _pending_abandoned_events' own comment for why this + // cannot be deferred to dump()-time). A no-op (dropped, counted via + // REFERENCE_CHAIN_EVENTS_DROPPED) if the queue is already at + // MAX_PENDING_ABANDONED_EVENTS. void enqueuePendingAbandonedEvent(); public: @@ -985,45 +2990,91 @@ class ReferenceChainTracker { } // tid -> java.lang.Thread global-ref registry (see _thread_objects). + // registerThreadObject() no-ops while !_enabled (Profiler::onThreadStart + // calls it unconditionally); unregisterThreadObject() is deliberately NOT + // gated on _enabled - a thread that started while enabled must release + // its global ref even after the recording stopped, otherwise the ref + // leaks for the JVM's lifetime. void registerThreadObject(JNIEnv *jni, int tid, jthread thread); void unregisterThreadObject(JNIEnv *jni, int tid); - // Delete the global refs unregisterThreadObject() queued in _thread_refs_pending_delete (see that - // member's comment for why the deletion is deferred). + // Delete the global refs unregisterThreadObject() queued in + // _thread_refs_pending_delete (see that member's comment for why the + // deletion is deferred). Called only from points where no walk phase + // holds a copied Thread-object ref: the BFS thread at the start of each + // pass, and Profiler::stop() after stopThread() has joined the BFS + // thread. A jni of null (current thread not attached) is a no-op. void releaseEndedThreadRefs(JNIEnv *jni); - // Recording-stop cleanup: delete EVERY remaining registered Thread global ref (dead threads' - // queued refs first, then the live registry) and empty the registry. + // Recording-stop cleanup: delete EVERY remaining registered Thread global + // ref (dead threads' queued refs first, then the live registry) and empty + // the registry. Only Profiler::stop() calls this, after stopThread() has + // joined the BFS thread (no walk phase can hold a copied ref) and with + // thread-event notifications about to be disabled - without it, refs of + // threads still alive at stop would never be released (nothing walks the + // registry again until a future recording re-registers a tid). A jni of + // null (current thread not attached) is a no-op. void releaseAllThreadObjects(JNIEnv *jni); - // One-time sweep over the JVM's CURRENTLY LIVE threads at recording start, registering each into - // the same tid -> Thread-object registry via JVMThread::nativeThreadId(). + // One-time sweep over the JVM's CURRENTLY LIVE threads at recording start, + // registering each into the same tid -> Thread-object registry via + // JVMThread::nativeThreadId(). Profiler::onThreadStart() cannot cover this + // population: those threads started before the recording began, and a + // leaking thread is typically among them (alive since process start). + // Called from Profiler::start(), not this class's own start(): gtest + // binaries call start() directly against partial mock JVMTI tables without + // a GetAllThreads slot, and only the real profiler lifecycle guarantees a + // fully populated JVMTI/JNI environment. void registerExistingThreads(jvmtiEnv *jvmti, JNIEnv *jni); - // Correlate a leak tag with an instance the BFS admitted BEFORE tagLeakInstances() tagged it (its - // JVMTI tag is a frontier tag, its frontier entry has leak_tag == 0). + // Correlate a leak tag with an instance the BFS admitted BEFORE + // tagLeakInstances() tagged it (its JVMTI tag is a frontier tag, its + // frontier entry has leak_tag == 0). Sets the entry's leak_tag so chain + // events emit targetTag = the leak tag (the HeapLiveObject correlation + // key), and records the instance as discovered. Returns false if the tag + // resolves to no live frontier entry (caller should treat the object as + // un-tagged). Idempotent: an entry that already carries a leak tag just + // returns true. Also handles post-restart re-admission: the new entry for + // a re-admitted instance gets the SAME leak tag the pool already holds + // for it (LivenessTracker's record survives the search restart). bool correlateAdmittedLeakTag(jlong frontier_tag, jlong leak_tag, u32 klass_id); - // Abandon the search after this many consecutive passes with zero new frontier entries admitted - // (genuinely stuck, not just slow). + // Abandon the search after this many consecutive passes with + // zero new frontier entries admitted (genuinely stuck, not just slow). + // A large heap takes more passes simply because there are more + // objects to explore — that is not "stuck". Only abandon when the + // frontier stops growing entirely. static constexpr int NO_PROGRESS_PASS_LIMIT = 30; - // Base limit for the canary-specific stuck detector: candidate-discovery must show no progress - // (no change to _candidate_found_bits, no new candidate admitted into a slot) for this many - // consecutive passes AND the whole-graph frontier must also have stalled for - // NO_PROGRESS_PASS_LIMIT passes (see runPass()'s CANARY_STUCK branch) before a canary search is - // abandoned. + // Base limit for the canary-specific stuck detector: candidate-discovery + // must show no progress (no change to _candidate_found_bits, no new + // candidate admitted into a slot) for this many consecutive passes AND + // the whole-graph frontier must also have stalled for NO_PROGRESS_PASS_LIMIT + // passes (see runPass()'s CANARY_STUCK branch) before a canary search is + // abandoned. The whole-graph requirement was added after live evidence + // showed the frontier still growing tens of thousands of entries deep + // while chasing a specific, confirmed-reachable candidate - a canary + // search is not "stuck" just because it hasn't found its candidate yet + // if the graph walk itself is still making real progress toward it. + // canaryStuckPassLimit() escalates this base value across consecutive + // CANARY_STUCK restarts of the same candidate-chase sequence (see + // _canary_stuck_restart_count), since a fixed cutoff cannot distinguish + // "genuinely unreachable within any reasonable budget" from "reachable, + // but deeper than one restart cycle can cover" - a large/deep heap + // legitimately needs more passes, not a smaller one. static constexpr int CANARY_NO_PROGRESS_PASS_LIMIT = 30; - // Upper bound on how many times canaryStuckPassLimit() doubles the base limit (2^8 = 256x -> 7680 - // passes at the default base of 30) - bounds the escalation so a search that is ACTUALLY stuck - // forever (as opposed to merely deep) still gets abandoned in finite time rather than growing its - // patience without limit. + // Upper bound on how many times canaryStuckPassLimit() doubles the base + // limit (2^8 = 256x -> 7680 passes at the default base of 30) - bounds + // the escalation so a search that is ACTUALLY stuck forever (as opposed + // to merely deep) still gets abandoned in finite time rather than + // growing its patience without limit. static constexpr int MAX_CANARY_STUCK_BACKOFF_SHIFT = 8; - // The canary-stuck pass limit for the *current* restart attempt: CANARY_NO_PROGRESS_PASS_LIMIT - // doubled once per consecutive CANARY_STUCK restart of this candidate-chase sequence, capped at + // The canary-stuck pass limit for the *current* restart attempt: + // CANARY_NO_PROGRESS_PASS_LIMIT doubled once per consecutive CANARY_STUCK + // restart of this candidate-chase sequence, capped at // MAX_CANARY_STUCK_BACKOFF_SHIFT doublings. int canaryStuckPassLimit() const { return CANARY_NO_PROGRESS_PASS_LIMIT @@ -1031,20 +3082,51 @@ class ReferenceChainTracker { MAX_CANARY_STUCK_BACKOFF_SHIFT); } - // Multiplier cap for the canary lane's work-scaled backoff (see _canary_backoff_mult's own - // comment). + // Multiplier cap for the canary lane's work-scaled backoff (see + // _canary_backoff_mult's own comment). 16 bounds a stuck chase's steady + // burn to ~1/16 of a core on pass work while keeping a deep-but-cheap + // (ms-scale passes) chase dense enough to resolve within a scenario's + // round window - sized against a real ~200-pass deep chase, which a + // fixed 1s cap starved outright (held-off wakes outpaced the run's + // window). Abandonment is not this knob's job + // (CANARY_STUCK's frontier-aware detector owns that); this only paces. static constexpr int CANARY_BACKOFF_MULT_MAX = 16; - // While a canary search has candidates still unresolved, shouldRunPass() raises - // _cpu_pain_budget's refill rate by this factor (capped at 100%/wall-clock). + // While a canary search has candidates still unresolved, shouldRunPass() + // raises _cpu_pain_budget's refill rate by this factor (capped at + // 100%/wall-clock). With the canary lane's rate now bounded by + // _canary_backoff_mult (above), this is no longer a rate control at all - + // it exists only so the conservative base refill tuned for the ordinary + // ~1 pass/s whole-graph cadence does not double-throttle a chase the + // backoff has already paced. The old covering (15x) vs emergency (100x) + // distinction is gone: both existed to feed the back-to-back mode the + // backoff replaces. static constexpr double CANARY_PAIN_BUDGET_REFILL_MULTIPLIER = 100.0; // Coverage tracking: how many leak tags have been assigned vs resolved. + // When all assigned tags are resolved (have chains), drop to 1x. int _leak_tags_assigned = 0; int _leak_tags_resolved = 0; - // Leak tags are positive JVMTI tags in a dedicated range, assigned by LivenessTracker's tag pool - // to specific tracked leaking objects. + // Base marker tag for canary-search candidates. RETIRED: pollWatchedTargets() + // stopped assigning marker tags when discovery moved to LivenessTracker's + // leak tags ("no marker tags - using leak tags now"); the constant and the + // _candidate_tags[] array remain only for the test seams and the dead- + // representative reconstruction path in pollWatchedTargets(). Historical + // rationale: each candidate i got MARKER_TAG_BASE - i (distinct negative + // values) so heapReferenceCallback() could tell which candidate was found; + // negative to avoid collision with frontier tags (positive, from _next_tag) + // and class tags (negative, from nextClassTag()/ClassTagAllocator, whose + // magnitudes are tiny next to 2^62). + static constexpr jlong MARKER_TAG_BASE = -(1LL << 62); + + // Leak tags are positive JVMTI tags in a dedicated range, assigned by + // LivenessTracker's tag pool to specific tracked leaking objects. The + // BFS recognizes them by range check and admits the object into the + // frontier, storing the leak tag in FrontierEntry::leak_tag for + // correlation with HeapLiveObject events. Unlike marker tags (one per + // candidate class), leak tags are per-instance — each tracked leaking + // object gets its own tag from a reusable pool. static constexpr jlong LEAK_TAG_BASE = 0x40000000LL; static constexpr int LEAK_TAG_POOL_SIZE = 256; @@ -1053,17 +3135,20 @@ class ReferenceChainTracker { return tag >= LEAK_TAG_BASE && tag < LEAK_TAG_BASE + LEAK_TAG_POOL_SIZE; } - // Max candidates LivenessTracker::selectLeakCandidates() can return. + // Max candidates LivenessTracker::selectLeakCandidates() can return. Must match + // LivenessTracker::MAX_LEAK_CANDIDATES. Duplicated here + // to avoid a heavy include chain (livenessTracker.h pulls jvmti.h). + // (Also declared in the private section above for field sizing.) // Test accessor for _passes_since_last_progress. int passesSinceLastProgressForTest() const { return _passes_since_last_progress; } // Canary-lane backoff state - see _canary_backoff_mult's own comment. int canaryBackoffMultForTest() const { return _canary_backoff_mult; } - u64 canaryPassEmaMsForTest() const { return _canary_pass_ema_ms; } + u64 canaryPassEmaNsForTest() const { return _canary_pass_ema_ns; } u64 lastCanaryPassNsForTest() const { return _last_canary_pass_ns; } - void setCanaryBackoffForTest(int mult, u64 ema_ms, u64 last_pass_ns) { + void setCanaryBackoffForTest(int mult, u64 ema_ns, u64 last_pass_ns) { _canary_backoff_mult = mult; - _canary_pass_ema_ms = ema_ms; + _canary_pass_ema_ns = ema_ns; _last_canary_pass_ns = last_pass_ns; } void setOomRampActiveForTest(bool active) { _oom_ramp_active = active; } @@ -1080,11 +3165,6 @@ class ReferenceChainTracker { } u64 candidateFoundBitsForTest() const { return _candidate_found_bits; } void setCandidateFrontierTagForTest(int idx, jlong tag) { _candidate_frontier_tags[idx] = tag; } - void setCandidateParentTagForTest(int idx, jlong tag) { _candidate_parent_tags[idx] = tag; } - void setCandidateReferrerKlassForTest(int idx, u32 klass_id) { - _candidate_referrer_klasses[idx] = klass_id; - } - void setCandidateDepthForTest(int idx, u32 depth) { _candidate_depths[idx] = depth; } int passesSinceLastCandidateProgressForTest() const { return _passes_since_last_candidate_progress; } int canaryStuckRestartCountForTest() const { return _canary_stuck_restart_count; } @@ -1093,20 +3173,34 @@ class ReferenceChainTracker { Error start(Arguments &args); - // Scales unset referencechains defaults (budget, ttl, framecap, pausetarget, painbudget, - // firstpassbudget) from the process's max heap size and available processor count, so a large - // heap doesn't starve the BFS (the defaults are tuned for a small heap and abandon via TTL before - // making meaningful progress). + // Scales unset referencechains defaults (budget, ttl, framecap, + // pausetarget, painbudget, firstpassbudget) from the process's max heap + // size and available processor count, so a large heap doesn't starve + // the BFS (the defaults are tuned for a small heap and abandon via + // TTL before making meaningful progress). Only overrides defaults + // that the operator did not set explicitly (tracked by + // args._reference_chains_tuned_mask). hop_cap is left alone: it bounds + // chain depth, not search breadth, and 200 is already generous. void autoTuneDefaults(Arguments &args); void stop(); - // Spawns the BFS thread (threadEntry()/threadLoop()) if reference chain tracking is enabled and - // no thread is already running. + // Spawns the BFS thread (threadEntry()/threadLoop()) if reference chain + // tracking is enabled and no thread is already running. Deliberately kept + // separate from start() itself: start() must stay safely callable with no + // live JVM attached (referenceChains_ut.cpp calls it directly against a + // mocked jvmtiEnv, with VM::_vm never set), while startThread()'s + // threadLoop() calls VM::attachThread() unconditionally - safe only once + // the JVM is actually up. Wired from Profiler::start() (profiler.cpp), + // which only calls this after the JVM/JVMTI environment is fully + // initialized, resolving the ordering concern start()'s own comment used + // to raise. No-op if disabled or already running. void startThread(); - // Stops and joins the BFS thread started by startThread(), mirroring BaseWallClock::stop()'s - // pthread_kill(WAKEUP_SIGNAL) + pthread_join() shape (wallClock.cpp) - WAKEUP_SIGNAL is already - // installed unconditionally in vmEntry.cpp, so no extra signal setup is needed here. + // Stops and joins the BFS thread started by startThread(), mirroring + // BaseWallClock::stop()'s pthread_kill(WAKEUP_SIGNAL) + pthread_join() + // shape (wallClock.cpp) - WAKEUP_SIGNAL is already installed + // unconditionally in vmEntry.cpp, so no extra signal setup is needed here. + // No-op if the thread was never started. void stopThread(); bool enabled() const { return _enabled; } @@ -1114,40 +3208,103 @@ class ReferenceChainTracker { u64 gcStartEpoch() { return load(_gc_start_epoch); } u64 gcFinishEpoch() { return load(_gc_finish_epoch); } - // JVMTI tag helpers used by the heap-walk callbacks. - jlong nextTag() { return atomicIncRelaxed(_next_tag, (jlong)1); } + // Tag round-trip helpers, reused by resolveLoadedClasses()/ + // heapReferenceCallback() (the heap-walk engine) to drive FrontierTable's tag-indexed + // slots. + // The frontier-tag namespace must stay disjoint from LivenessTracker's + // leak-tag range [0x40000000, 0x40000100) (LEAK_TAG_BASE, + // livenessTracker.h - duplicated here rather than included, per this + // class's no-livenessTracker-include rule; the test suite asserts the two + // constants agree). isLeakTag() classifies by magnitude alone. + // _next_tag is an unbounded incrementing counter reset only by + // restartSearch(); a single search admitting >= 2^30 objects (possible on + // a very large heap with a high frontier cap and no restart) would walk + // _next_tag into the leak range and misclassify fresh frontier tags as + // leak tags, corrupting isLeakTag()-dispatched handling in + // heapReferenceCallback(). Guard the boundary: at the cap the search is + // forced to abandon (restartSearch() rewinds _next_tag to 1), the same + // outcome any other search-bound exhaustion produces. In debug builds the + // assert makes the invariant checkable in tests. + static constexpr jlong FRONTIER_TAG_NAMESPACE_CEILING = 0x40000000LL; + jlong nextTag() { + jlong tag = atomicIncRelaxed(_next_tag, (jlong)1); + if (tag >= FRONTIER_TAG_NAMESPACE_CEILING) { + // Never hand out a tag inside the leak-tag namespace: ask the walk to + // abandon (the abort flag is checked in both heap callbacks) and + // return 0, which every caller's insert() rejects - the same outcome + // as any other search-bound exhaustion. + _abort_pass_requested.store(true, std::memory_order_relaxed); + return 0; + } + assert(tag > 0 && "frontier tags are positive"); + return tag; + } - // Serializes runPass()+pollWatchedTargets() between threadLoop() and the test seams - see - // runPassForTest()'s comment. + // Serializes runPass()+pollWatchedTargets() between threadLoop() and the + // test seams - see runPassForTest()'s comment. A full pthread mutex, not + // a spin lock: the critical section is a whole BFS pass (tens of ms), far + // too long to spin, and neither holder is ever a signal context. Mutex _engine_lock; jlong tagObject(jvmtiEnv *jvmti, jobject obj); jlong getTag(jvmtiEnv *jvmti, jobject obj); void clearTag(jvmtiEnv *jvmti, jobject obj); - // Hands out a fresh negative class tag, from the shared, process-wide counter both this class and - // LivenessTracker mint from - see classTagAllocator.h's own header comment for why this must be - // shared rather than a private counter here. + // Hands out a fresh negative class tag, from the shared, process-wide + // counter both this class and LivenessTracker mint from - see + // classTagAllocator.h's own header comment for why this must be shared + // rather than a private counter here. Exposed (not just used internally + // by resolveLoadedClasses()) so tests can drive class tagging directly + // against a mocked jvmtiEnv without going through GetLoadedClasses. jlong nextClassTag() { return ClassTagAllocator::next(); } - // Returns the frontier metadata table, or nullptr if the subsystem was never started with the - // flag enabled. + // Returns the frontier metadata table, or nullptr if the subsystem was + // never started with the flag enabled. FrontierTable *frontierTable() { return _frontier; } - // Returns the class-tag resolution table. Exposed for testing in isolation, matching - // frontierTable()'s existing rationale. + // Returns the class-tag resolution table. Exposed for testing in + // isolation, matching frontierTable()'s existing rationale. ClassTagTable *classTags() { return &_class_tags; } - // Runs exactly one bounded BFS pass and returns. The first call for a search seeds - // FollowReferences from the heap roots (heap_filter=0, klass=NULL, initial_object=NULL - see this - // method's own comment in referenceChains.cpp for why FollowReferences rather than - // IterateThroughHeap); every later call resumes from the persisted frontier via expandFrontier() - // instead of re-walking from the roots (see expandFrontier()'s comment for why - re-walking from - // the roots each call would re-traverse the entire already-discovered subgraph every pass, - // defeating the point of a per-pass budget). + // Runs exactly one bounded BFS pass and returns. The first call for a + // search seeds FollowReferences from the heap roots (heap_filter=0, + // klass=NULL, initial_object=NULL - see this method's own comment in + // referenceChains.cpp for why FollowReferences rather than + // IterateThroughHeap); every later call resumes from the persisted + // frontier via expandFrontier() instead of re-walking from the roots (see + // expandFrontier()'s comment for why - re-walking from the roots each call + // would re-traverse the entire already-discovered subgraph every pass, + // defeating the point of a per-pass budget). Newly discovered objects are + // admitted into frontierTable() up to _hop_cap/_budget/the frontier + // table's own capacity cap. + // + // Returns false if reference chain tracking is disabled, jvmti is null, or + // the frontier table was never constructed (start() never ran with the + // flag enabled). A pass that hits its budget/hop/frontier cap is still a + // *successful* call (returns true) - *out_truncated (if non-null) reports + // whether *this pass* ran to full exhaustion of the currently-known + // reachable graph or was cut short, per the design doc's "no silent + // truncation" requirement; this is call-scoped, unlike searchState() + // below which reports the whole search's outcome. + // + // Once searchState() is no longer RUNNING (the reachable graph was fully + // explored within caps, or the search was abandoned - see the Termination + // section implemented below), further calls are no-ops that return true + // immediately, *unless* shouldRunPass() has already called restartSearch() + // to begin a fresh search (this class's own header comment) - in that case + // _search_started is false again and this method takes the first-pass + // branch exactly as it would for a brand-new tracker. bool runPass(jvmtiEnv *jvmti, JNIEnv *jni, bool *out_truncated = nullptr); - // Serialized entry points for the two engine drivers: the real BFS thread (threadLoop(), below) - // and the debug seams (javaApi.cpp's runReferenceChainPass0()/pollReferenceChainTargets0()). + // Serialized entry points for the two engine drivers: the real BFS thread + // (threadLoop(), below) and the debug seams (javaApi.cpp's + // runReferenceChainPass0()/pollReferenceChainTargets0()). The engine's + // non-frontier maps (_class_tags, _candidate_*, _leak_parent_fanout, ...) + // are plain containers with no cross-thread locking, so a seam-driven + // pass on a test thread while threadLoop() is mid-pass is a genuine data + // race - observed: SIGSEGV in ClassTagTable::insert's unordered_map + // rehash from a test thread inside resolveLoadedClasses() while the BFS + // thread was mid-pass of its own. Taking _engine_lock at both entry + // points makes the two drivers mutually exclusive while either can run. bool runPassSerialized(jvmtiEnv *jvmti, JNIEnv *jni) { MutexLocker engine_guard(_engine_lock); return runPass(jvmti, jni); @@ -1158,43 +3315,91 @@ class ReferenceChainTracker { pollWatchedTargets(jvmti, jni); } - // Search-level outcome (SearchState's constants) - see runPass()'s comment for exactly when this - // leaves RUNNING. + // Search-level outcome (SearchState's constants) - see runPass()'s comment + // for exactly when this leaves RUNNING. Acquire-loaded, pairing with + // runPass()'s release store of this same field (referenceChains.cpp), so a + // caller that observes a non-RUNNING value here also sees every detail + // field (_abandon_reason, _passes_run, ...) runPass() wrote before that + // release store. u8 searchState() { return loadAcquire(_search_state); } - // Total passes run for the current/most recent search. Exposed for tests to confirm multi-pass - // resumption actually happened. + // Total passes run for the current/most recent search. Exposed for tests + // to confirm multi-pass resumption actually happened. int passesRun() { return load(_passes_run); } - // Which SearchAbandonReason cutoff moved the search out of RUNNING, or SearchAbandonReason::NONE - // if it never left RUNNING or left via SearchState::COMPLETED instead. + // Which SearchAbandonReason cutoff moved the search out of RUNNING, or + // SearchAbandonReason::NONE if it never left RUNNING or left via + // SearchState::COMPLETED instead. u8 abandonReason() { return load(_abandon_reason); } // Reference-chain JFR event surface: fills *out from frontierTable()-> - // reconstructChain(target_tag, ...) (see that method's own comment for the leaf-to-root ordering - // and the parent_tag walk it performs). + // reconstructChain(target_tag, ...) (see that method's own comment for + // the leaf-to-root ordering and the parent_tag walk it performs). Returns + // false (leaving *out untouched) if target_tag was never inserted into + // the frontier table - the same failure case reconstructChain() itself + // reports, just wrapped into the JFR-event shape + // Recording::recordReferenceChain() (flightRecorder.cpp) expects. + // + // Deliberately does not decide *when* to call this or *which* target_tag + // to use - this codebase has no target-sample feed into + // ReferenceChainTracker yet (see runPass()'s own comment), so wiring an + // automatic call site here would have to invent + // that feed rather than reuse one. A future consumer that knows which + // tag it is chasing (e.g. an ObjectSampler-driven target) calls this + // directly once that feed exists. + // Resolves one chain hop's retention-edge label into `out` (NUL- + // terminated, at most out_cap bytes incl. the NUL). For a FIELD/STATIC_FIELD + // edge with a resolvable referrer class, the label is the field's NAME, + // decoded per the JVMTI specification's field-ordinal scheme (see + // FrontierEntry::referrer_field_index's own comment - the full scheme, with + // the interface-branch and class-branch rules of jvmtiHeapReferenceInfoField). + // Any other edge kind, or any decode failure (class gone, ordinal out of + // the computed range, JVMTI call failure), degrades to the edge KIND label + // ("element", "constant_pool", ...) - never a fabricated name: a wrong + // numbering on an unverified JVM would silently degrade to kind labels, + // not lie. HotSpot's numbering is source-verified (jvmtiTagMap.cpp + // ClassFieldMap::create_map_of_static_fields/instance_fields build exactly + // the spec's ordinal space); J9's compliance is INFERRED from the spec + // definition only (its walker arithmetic not yet source-verified - see + // find-field-name-decoding node). + // Legal only OUTSIDE heap callbacks (FindClass/GetClassFields/GetFieldName + // are not callable during heap iteration) - buildChainEvent() calls it on + // the BFS thread between walks. void resolveHopEdgeLabel(jvmtiEnv *jvmti, JNIEnv *jni, ChainHopEdge edge, char *out, size_t out_cap); - // Fills *out with one label per chain hop, aligned with the chain's leaf-to-root order (edges[i] - // = the retention edge INTO chain[i]), via resolveHopEdgeLabel() above. + // Fills *out with one label per chain hop, aligned with the chain's + // leaf-to-root order (edges[i] = the retention edge INTO chain[i]), via + // resolveHopEdgeLabel() above. Truncates labels to + // MAX_REFERENCE_CHAIN_EDGE_LABEL (event.h) bytes. Safe with null jvmti/jni + // (every label degrades to the edge kind) and against partial JVMTI + // function tables (gtest mock environments - the required slots are + // null-checked before use, same defensive rule the JFR-roundtrip crash + // taught for RCT::start()). static constexpr size_t MAX_HOP_EDGE_LABEL = MAX_REFERENCE_CHAIN_EDGE_LABEL; - // Per-referrer-class ordinal->name list cache behind resolveHopEdgeLabel(): decoding the spec - // ordinal requires walking the class's whole interface closure + superclass chain (GetClassFields - // + GetFieldName per field), and chains re-emit on every dump, so the decoded ordinal space of - // each chain-relevant class is built once here. + // Per-referrer-class ordinal->name list cache behind + // resolveHopEdgeLabel(): decoding the spec ordinal requires walking the + // class's whole interface closure + superclass chain (GetClassFields + + // GetFieldName per field), and chains re-emit on every dump, so the decoded + // ordinal space of each chain-relevant class is built once here. Keyed by + // the referrer's raw class tag (stable per loaded class). Bounded by + // HOP_LABEL_CLASS_CACHE_CAP (cleared whole on search restart - the frontier + // and its class tags do not survive a restart, so nothing here may). static constexpr size_t HOP_LABEL_CLASS_CACHE_CAP = 1024; struct HopLabelClass { jlong class_tag; - // One entry per ordinal in the class's flattened field space - the i-th element is the name of - // ordinal i. + // One entry per ordinal in the class's flattened field space - the + // i-th element is the name of ordinal i. Empty when decoding failed + // (all hops through this class degrade to kind labels). std::vector field_names; bool decode_failed; }; std::unordered_map _hop_label_cache; - // Cache lookup/decode behind resolveHopEdgeLabel() - see HopLabelClass's own comment. + // Cache lookup/decode behind resolveHopEdgeLabel() - see HopLabelClass's + // own comment. Never returns null (a failed decode is cached as + // decode_failed and re-reported as kind labels). const HopLabelClass *hopLabelClassFor(jvmtiEnv *jvmti, JNIEnv *jni, jlong class_tag); @@ -1206,21 +3411,38 @@ class ReferenceChainTracker { ReferenceChainEvent *out); // Appends the root TYPE element (the declaring class, resolved from - // FrontierEntry::referrer_class_tag) to a static-field-rooted chain - see the definition's - // comment in referenceChains.cpp for the full rationale and the skip conditions. + // FrontierEntry::referrer_class_tag) to a static-field-rooted chain - see + // the definition's comment in referenceChains.cpp for the full rationale + // and the skip conditions. void appendStaticFieldRootType(const FrontierEntry &terminal, std::vector *chain, std::vector *edges); - // Canary-search chain reconstruction: builds the chain for a canary candidate from the - // per-candidate chain link recorded at pruning time (_candidate_parent_tags[] etc.), walking - // parent_tag through the frontier table (positive tags, so lookup() works). + // Canary-search chain reconstruction: builds the chain for a canary + // candidate from the per-candidate chain link recorded at + // pruning time (_candidate_parent_tags[] etc.), walking + // parent_tag through the frontier table (positive tags, + // so lookup() works). The candidate's own + // referrer_klass is prepended to the chain. bool buildCanaryChainEvent(int candidate_idx, ReferenceChainEvent *out); - // Reports a search's termination state without needing a target tag. + // Abandoned-search JFR event surface for the design doc's "explicit reporting of + // abandoned searches" requirement - unlike buildChainEvent() above, this + // needs no target_tag: it reports the search's own termination state, + // which runPass() (referenceChains.cpp) already tracks unconditionally. + // Returns false (leaving *out untouched) if the search was never + // abandoned (searchState() != SearchState::ABANDONED). Same-thread only: + // the sole caller is enqueuePendingAbandonedEvent() (referenceChains.cpp), + // running on the BFS thread immediately after runPass() writes + // SearchState::ABANDONED - the queueing exists precisely because a + // cross-thread live read from Profiler::dump() raced the next restart + // (see _pending_abandoned_events' own comment). The plain reads of + // _hop_cap/_budget/_ttl_ms/_frontier are therefore same-thread with their + // writers (start()/stop() configuration and the BFS thread), not the + // cross-thread reads an earlier draft of this comment described. bool buildAbandonedEvent(ReferenceChainAbandonedEvent *out) { - // Acquire-load, not a plain relaxed load - see searchState()'s own comment for why: this is the - // same guard-then-read-details pattern. + // Acquire-load, not a plain relaxed load - see searchState()'s own + // comment for why: this is the same guard-then-read-details pattern. if (out == nullptr || loadAcquire(_search_state) != SearchState::ABANDONED) { return false; } @@ -1234,65 +3456,132 @@ class ReferenceChainTracker { return true; } - // Bridges LivenessTracker leak candidates into cached ReferenceChain events; reads the existing - // tag, never seeds one. + // Target-selection bridging step (design doc's Open Question 3, corrected + // mechanism - see this class's own header comment's bridging-step note): + // polls + // LivenessTracker::selectLeakCandidates() and, for each candidate whose + // representative instance has already been discovered by an ordinary + // runPass() walk (getTag() > 0 - a read, never a SetTag seed), reconstructs + // its datadog.ReferenceChain and caches it in _resolved_chains keyed by + // klass_id - see that field's own comment for why a resolved chain is + // cached (and re-emitted on every dump) rather than emitted once. The write + // itself is still deferred to drainPendingChainEvents() on the dump() + // thread, since the profiler-side writer can block this method's + // caller (the BFS scheduling thread) for up to ~50ms per event. A candidate + // still at tag 0 (not yet discovered) is left for a later poll to retry, + // since runPass()'s whole-graph walk eventually visits every root-reachable + // object, barring the hop/budget/frontier caps. A klass already cached from + // the current search generation is not reconstructed again; a restart + // (new _search_start_ns) or a re-tag makes the next poll refresh it. Every + // cached entry whose representative no longer resolves (collected/evicted) + // is pruned here, so the cache tracks the set of still-live flagged samples. + // Called from threadLoop() once per scheduling cycle, after runPass(), so + // this poll always sees the most recent pass's tagging. No-op if + // disabled, or if jvmti/jni is null (mirrors runPass()'s own null-safety, + // so a test can call this directly without a live JVM attached, the same + // way referenceChains_ut.cpp already does for runPass()). void pollWatchedTargets(jvmtiEnv *jvmti, JNIEnv *jni); - // Targeted holder re-walk: enqueues `tag`'s chain-root entry (the root-attached ancestor of its - // frontier chain) onto _priority_expand so the next rotation/expand pass re-walks the holder that - // retains everything below `tag`. + // Targeted holder re-walk: enqueues `tag`'s chain-root entry (the + // root-attached ancestor of its frontier chain) onto _priority_expand so + // the next rotation/expand pass re-walks the holder that retains + // everything below `tag`. Rationale (observed live in the correlation + // scenario): a container that replaces its internals (growing ArrayList, + // resized HashMap) never appears in the fanout as the direct parent of + // anything watched - the watched instances' direct parents are the DEAD + // old internals - and the blind lap over a large frontier is far too slow + // to reach the holder in any realistic window, so new internals are never + // admitted and tagged leak instances below them are never intercepted. + // The holder chain's root, however, is exactly what a candidate's chain + // reconstruction already walks; requeueing it per poll (bounded by + // MAX_LEAK_CANDIDATES pushes, de-duplicated by isQueuedForRotation) makes + // the holder's CURRENT children - including each new backing array - + // admitted promptly. See _leak_parent_fanout's own comment for the + // complementary (probabilistic) ancestor coverage. void requeueChainRootForRotation(jlong tag); - // Appends a copy of every currently-cached resolved chain to *out, re-stamped with a fresh - // _start_time so it lands in the dumping chunk's time window, WITHOUT clearing the cache - a - // repeatable snapshot, not a drain, so the same live sample's chain is re-emitted into every JFR - // chunk it survives into (see _resolved_chains' own comment). + // Appends a copy of every currently-cached resolved chain to *out, + // re-stamped with a fresh _start_time so it lands in the dumping chunk's + // time window, WITHOUT clearing the cache - a repeatable snapshot, not a + // drain, so the same live sample's chain is re-emitted into every JFR chunk + // it survives into (see _resolved_chains' own comment). Called from + // Profiler::dump() (profiler.cpp), which then hands + // each event to the profiler-side writer on its own thread - never + // the BFS scheduling thread. A no-op (leaves *out untouched) if the cache + // is currently empty. The name is retained from the drain-once era for its + // stable call site; the semantics are now snapshot-and-keep. void drainPendingChainEvents(std::vector *out); - // Appends every abandoned-search event queued since the last call and clears the queue - a true - // drain, unlike drainPendingChainEvents() above: an abandoned search is a discrete past - // occurrence, not an ongoing live sample, so there is nothing left to re-report once - // Profiler::dump() (profiler.cpp) has emitted it. + // Appends every abandoned-search event queued since the last call and + // clears the queue - a true drain, unlike drainPendingChainEvents() above: + // an abandoned search is a discrete past occurrence, not an ongoing live + // sample, so there is nothing left to re-report once Profiler::dump() + // (profiler.cpp) has emitted it. Exists because searchState()/ + // buildAbandonedEvent() alone cannot be read reliably from dump()'s thread + // (see _pending_abandoned_events' own comment): each event here was + // snapshotted synchronously, on the BFS thread, at the exact moment the + // search abandoned - before shouldRunPass() gets a chance to call + // restartSearch() and clear the live fields buildAbandonedEvent() would + // otherwise have read. void drainPendingAbandonedEvents(std::vector *out); static void JNICALL GarbageCollectionStart(jvmtiEnv *jvmti_env); static void JNICALL GarbageCollectionFinish(jvmtiEnv *jvmti_env); - // Test seam - not part of the production API. Mirrors LivenessTracker's own "Test seams" block - // (livenessTracker.h). + // Test seam - not part of the production API. Mirrors LivenessTracker's + // own "Test seams" block (livenessTracker.h). Production code only ever + // discovers frontier roots via runPass()'s root-seeded FollowReferences + // walk; this lets a test tag and insert one specific, caller-chosen live + // object as a frontier root directly, so runPass()/pollWatchedTargets()/ + // buildChainEvent() can be exercised end-to-end against a known target + // without depending on LivenessTracker's probabilistic allocation sampler + // to organically select and surface the same object. Returns the assigned + // tag (matching the value buildChainEvent()'s target_tag expects), or 0 on + // failure (obj/jvmti/jni null, SetTag failed, or the frontier table is at + // capacity). jlong tagAsRootForTest(jvmtiEnv *jvmti, JNIEnv *jni, jobject obj); - // Test seam - not part of the production API. Since ReferenceChainTracker is a process-wide - // singleton (ExternalProcessReferenceChainTest's own class javadoc explains why that matters: - // only the *first* test to ever call runPass() in a shared JVM gets a real root-seeded walk, - // since runPass() only re-walks from the roots once per search's whole lifetime), an in-process - // test that needs its own genuine first-ever root walk calls this at the start of its test body - // to force exactly that - releasing any tags a previous test's search still held, then resetting - // search/frontier state to the same "brand-new tracker" state restartSearch() - // (referenceChains.cpp) produces, plus the target- dedup/pending-event state restartSearch() - // itself intentionally leaves for pollWatchedTargets()/drainPendingChainEvents() to self-clear - // (this is an immediate, out-of-band reset - there is no next real pass here to observe the - // change and clear them the ordinary way). + // Test seam - not part of the production API. Since ReferenceChainTracker + // is a process-wide singleton (ExternalProcessReferenceChainTest's own + // class javadoc explains why that matters: only the *first* test to ever + // call runPass() in a shared JVM gets a real root-seeded walk, since + // runPass() only re-walks from the roots once per search's whole + // lifetime), an in-process test that needs its own genuine first-ever + // root walk calls this at the start of its test body to force exactly + // that - releasing any tags a previous test's search still held, then + // resetting search/frontier state to the same "brand-new tracker" state + // restartSearch() (referenceChains.cpp) produces, plus the target- + // dedup/pending-event state restartSearch() itself intentionally leaves + // for pollWatchedTargets()/drainPendingChainEvents() to self-clear (this + // is an immediate, out-of-band reset - there is no next real pass here to + // observe the change and clear them the ordinary way). void resetSearchStateForTest(jvmtiEnv *jvmti, JNIEnv *jni); - // Test seam - not part of the production API. Diagnostic-only: reports how far a given - // (already-tagged) object sits from the front of _pending_expand's FIFO queue, to distinguish - // "not yet expanded because its own FIFO position hasn't come up yet" from "already expanded" or - // "never admitted at all" without needing a debugger. + // Test seam - not part of the production API. Diagnostic-only: reports how + // far a given (already-tagged) object sits from the front of + // _pending_expand's FIFO queue, to distinguish "not yet expanded because + // its own FIFO position hasn't come up yet" from "already expanded" or + // "never admitted at all" without needing a debugger. Returns >=0 (the + // 0-based distance from the front - 0 means it expands next) if tag is + // still queued, -1 if tag is nonzero but not currently queued (already + // expanded, or never admitted), or -2 if tag itself is 0. long pendingExpandPositionForTest(jlong tag) const; - // Test seam - not part of the production API. Companion to pendingExpandPositionForTest() above, - // for computing a position's fraction of the current backlog. + // Test seam - not part of the production API. Companion to + // pendingExpandPositionForTest() above, for computing a position's + // fraction of the current backlog. size_t pendingExpandSizeForTest() const; - // Test seam - not part of the production API. Exposes the private shouldRunPass() gate directly, - // so a test can assert whether a fresh/terminal search would be allowed to start right now - in - // particular, whether LivenessTracker::secondsToOOM()'s urgent-OOM bypass (hasLeakSignal(), see - // OOM_URGENT_THRESHOLD_S's own comment above) opens this gate even with zero per-klass leak - // candidate (confirmable in the same test via - // LivenessTracker::selectLeakCandidates()/JavaProfiler's selectLeakCandidateKlassIds0() seam) - - // something runReferenceChainPass0() (javaApi.cpp) cannot show, since it calls runPass() directly - // and never consults this gate at all. + // Test seam - not part of the production API. Exposes the private + // shouldRunPass() gate directly, so a test can assert whether a + // fresh/terminal search would be allowed to start right now - in + // particular, whether LivenessTracker::secondsToOOM()'s urgent-OOM bypass + // (hasLeakSignal(), see OOM_URGENT_THRESHOLD_S's own comment above) opens + // this gate even with zero per-klass leak candidate (confirmable in the + // same test via LivenessTracker::selectLeakCandidates()/JavaProfiler's + // selectLeakCandidateKlassIds0() seam) - something runReferenceChainPass0() + // (javaApi.cpp) cannot show, since it calls runPass() directly and never + // consults this gate at all. bool shouldRunPassForTest(u64 now_ns) { return shouldRunPass(now_ns); } }; diff --git a/ddprof-lib/src/main/cpp/vmEntry.cpp b/ddprof-lib/src/main/cpp/vmEntry.cpp index acf8382322..be3f5fa57e 100644 --- a/ddprof-lib/src/main/cpp/vmEntry.cpp +++ b/ddprof-lib/src/main/cpp/vmEntry.cpp @@ -395,6 +395,15 @@ bool VM::initLibrary(JavaVM *vm) { return true; } +// jvmtiEventCallbacks has a single function-pointer slot per event; both +// LivenessTracker and ReferenceChainTracker need GarbageCollectionFinish, +// so this trampoline dispatches to both instead of one +// subsystem's registration clobbering the other's. +static void JNICALL onGarbageCollectionFinish(jvmtiEnv *jvmti_env) { + LivenessTracker::GarbageCollectionFinish(jvmti_env); + ReferenceChainTracker::GarbageCollectionFinish(jvmti_env); +} + void VM::probeJFRRequestStackTrace() { jint ext_count = 0; jvmtiExtensionFunctionInfo *ext_functions = nullptr; @@ -445,17 +454,6 @@ bool VM::initializeRequestStackTrace() { return false; } -// JVMTI delivers ONE callback per event slot; both trackers need the -// GarbageCollectionFinish signal (liveness GC epochs + reference-chain pass -// scheduling), so this vmEntry-level forwarder fans the single slot out to -// both static callbacks. Order is irrelevant (each only does lock-free -// bookkeeping); LivenessTracker's callback keeps its own -// initCurrentThreadSignalSafe() behavior. -void JNICALL ForwardedGarbageCollectionFinish(jvmtiEnv *jvmti_env) { - LivenessTracker::GarbageCollectionFinish(jvmti_env); - ReferenceChainTracker::GarbageCollectionFinish(jvmti_env); -} - bool VM::initProfilerBridge(JavaVM *vm, bool attach) { TEST_LOG("VM::initProfilerBridge"); if (!initShared(vm)) { @@ -531,7 +529,7 @@ bool VM::initProfilerBridge(JavaVM *vm, bool attach) { callbacks.ThreadEnd = Profiler::ThreadEnd; callbacks.SampledObjectAlloc = ObjectSampler::SampledObjectAlloc; callbacks.GarbageCollectionStart = ReferenceChainTracker::GarbageCollectionStart; - callbacks.GarbageCollectionFinish = ForwardedGarbageCollectionFinish; + callbacks.GarbageCollectionFinish = onGarbageCollectionFinish; callbacks.NativeMethodBind = VMStructs::NativeMethodBind; _jvmti->SetEventCallbacks(&callbacks, sizeof(callbacks)); diff --git a/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java b/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java index c65d80b00d..dea1dee574 100644 --- a/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java +++ b/ddprof-lib/src/main/java/com/datadoghq/profiler/JavaProfiler.java @@ -462,6 +462,19 @@ public Map getDebugCounters() { private static native int getTid0(); + /** + * Test seam (debug native builds only): returns the calling thread's profiler tid + * (ProfiledThread::currentTid()'s value on the native side - the same tid space + * {@code seedTidTrendSample0} matches tracked instances' allocating threads + * against). Scenarios seeding per-(klass, tid) trend ramps must call this ON + * the leaking thread and pass the result to {@code seedTidTrendSample0}, or + * the qualifying tid will match no tracked instance and no leak tag will + * ever be assigned to the scenario's real instances. + */ + public static int getTid() { + return getTid0(); + } + private static native boolean recordTrace0(long rootSpanId, String endpoint, String operation, int sizeLimit); private static native void dump0(String recordingFilePath); @@ -530,6 +543,181 @@ private static native void setTraceContext0(long localRootSpanId, long spanId, l */ public static native boolean testTlsPrimingAvailable(); + /** + * Test seam (debug native builds only - a no-op returning {@code false}/{@code 0}/an + * empty array in release builds): decouples LivenessTracker's leak-candidate + * detection from ReferenceChainTracker's chain reconstruction, each independently + * verifiable end-to-end without depending on both the probabilistic JVMTI heap + * sampler and the reference-chain BFS search organically producing the right + * conditions in the same test run. + *

+ * Enables/disables LivenessTracker's per-klass population tracking directly, + * bypassing {@code initialize()}'s live-JVM requirement. Returns {@code true} on + * debug builds. + */ + static native boolean setGcGenerationsEnabled0(boolean enabled); + + /** + * Test seam (debug native builds only): seeds one epoch's worth of population + * history for {@code klassId} directly into LivenessTracker's ring buffer, + * bypassing real allocation sampling. Repeated calls (with distinct + * {@code epoch} values) build up a trend {@link #selectLeakCandidateKlassIds0()} + * can then rank, letting a test assert a slope signal would be generated for a + * chosen klass id without waiting on real GC epochs. + */ + static native void seedKlassPopulationSample0(int klassId, int count, long epoch); + + /** + * Test seam (debug native builds only): seeds one epoch's worth of per-tid trend + * history for {@code klassId}, the (klass, tid) qualification + * selectLeakCandidates() now requires on top of the klass-level ramp seeded by + * {@link #seedKlassPopulationSample0}. Repeated calls with the same rising shape + * (and the same {@code tid}) build the sustained rise that qualifies {@code tid} + * as a leak-site thread. {@code tid} must be the leaking thread's real profiler + * tid from {@link #getTid()} whenever the scenario also relies on leak tagging + * of real tracked instances; see that method's own comment. + */ + static native void seedTidTrendSample0(int klassId, int tid, int count, long epoch); + + /** + * Test seam (debug native builds only): wires {@code representative} in as {@code klassId}'s + * leak-candidate representative directly (a fresh weak global ref owned by LivenessTracker), + * bypassing the real allocation-sampling path that would otherwise populate this. Combined + * with {@link #seedKlassPopulationSample0} and {@link #tagAsReferenceChainRoot0}, lets a test + * join a synthetic slope signal to a real, directly-tagged object so + * {@link #pollReferenceChainTargets0()}'s bridging step can be exercised end-to-end with + * neither the real sampler nor the real root-seeded walk involved. + */ + static native void setKlassPopulationRepresentativeForTest0(int klassId, Object representative); + + /** + * Test seam (debug native builds only): clears LivenessTracker's per-klass population table, + * so a later test in the same JVM does not observe leak candidates seeded by an earlier one. + */ + static native void resetKlassPopulationForTest0(); + + /** + * Test seam (debug native builds only): returns the klass ids LivenessTracker's + * real leak-candidate ranking (positive population slope, top 5) currently + * selects - the same call ReferenceChainTracker's restart gate and target-polling + * bridge use in production, exposed here so a test can assert a slope signal was + * generated (real or seeded via {@link #seedKlassPopulationSample0}) without + * needing a reference-chain search to also be running. + */ + static native int[] selectLeakCandidateKlassIds0(); + + /** + * Test seam (debug native builds only): tags {@code target} and inserts it + * directly as a reference-chain frontier root, bypassing ReferenceChainTracker's + * normal discovery path (a root-seeded FollowReferences walk) and + * LivenessTracker's leak-candidate selection entirely. Lets a test drive + * {@link #runReferenceChainPass0()}/{@link #pollReferenceChainTargets0()} against + * a known, caller-chosen live object. Returns the assigned frontier tag (matching + * the {@code target_tag} a resulting {@code datadog.ReferenceChain} event + * reports), or {@code 0} on failure (reference chains disabled, or the frontier + * table is at capacity). + */ + static native long tagAsReferenceChainRoot0(Object target); + + /** + * Test seam (debug native builds only): runs exactly one bounded BFS pass of the + * reference-chain search synchronously, rather than waiting on the tracker's own + * background thread/cadence. Returns {@code false} if reference chains are + * disabled or the tracker was never started. + */ + static native boolean runReferenceChainPass0(); + + /** + * Test seam (debug native builds only): runs one poll of + * ReferenceChainTracker's LivenessTracker-to-chain-reconstruction bridging step + * synchronously - for each current leak candidate already discovered by a prior + * {@link #runReferenceChainPass0()} walk, reconstructs and queues its chain + * event, rather than waiting on the background thread's own scheduling cycle. + */ + static native void pollReferenceChainTargets0(); + + /** + * Test seam (debug native builds only): drains and returns the number of + * reference-chain events queued by {@link #pollReferenceChainTargets0()} so far + * (the same queue {@code Profiler.dump()} drains in production to write + * {@code datadog.ReferenceChain} JFR events) - lets a test assert a chain was + * actually reconstructed without needing a real JFR dump. + */ + static native int drainReferenceChainEventCount0(); + + /** + * Test seam (debug native builds only): resets ReferenceChainTracker's search/frontier state + * back to a brand-new tracker's, releasing any tags a previous search still held. Since the + * tracker is a process-wide singleton, an in-process test that needs its own genuine first + * root-seeded walk (runPass() only re-walks from the roots once per search's whole lifetime) + * calls this at the start of its test body to force one, rather than depending on being the + * first reference-chain test to run in a shared test JVM. + */ + static native void resetReferenceChainSearchForTest0(); + + /** + * Test seam (debug native builds only): diagnostic-only, does not tag {@code target}. Reads + * target's existing JVMTI tag (0 if the real search has never admitted it) and reports its + * FIFO distance from the front of ReferenceChainTracker's pending-expansion queue: {@code >=0} + * (0 = expands next) if still queued, {@code -1} if tagged but no longer queued (already + * expanded), or {@code -2} if never admitted at all. + */ + static native long getReferenceChainPendingPositionForTest0(Object target); + + /** + * Test seam (debug native builds only): the current size of ReferenceChainTracker's + * pending-expansion queue, for computing {@link #getReferenceChainPendingPositionForTest0}'s + * position as a fraction of the current backlog. + */ + static native long getReferenceChainPendingSizeForTest0(); + + /** + * Test seam (debug native builds only): seeds one heap-floor-ring sample - the input to + * {@code LivenessTracker::secondsToOOM()}'s time-to-OOM projection - directly, bypassing the + * real {@code GarbageCollectionFinish} callback. {@code timestampNs} values are only ever + * compared against each other, never against a real wall clock, so a test may use any + * self-consistent, strictly increasing sequence to build an arbitrary rising or flat + * heap-usage-over-time history without waiting on real GCs. + */ + static native void heapFloorRecordForTest0(long usedBytes, long timestampNs); + + /** + * Test seam (debug native builds only): overrides the max-heap-size {@code secondsToOOM()} + * projects against, bypassing the real {@code Runtime.maxMemory()} resolution - so a test can + * exercise the projection deterministically, independent of whatever {@code -Xmx} this JVM's + * own shared, no-{@code forkEvery} fork happens to run with. + */ + static native void setMaxHeapBytesForTest0(long maxHeapBytes); + + /** + * Test seam (debug native builds only): temporarily disables {@code onGC()}'s own + * {@code recordHeapFloorSample()} call so a test can seed the heap-floor ring exclusively + * via {@link #heapFloorRecordForTest0(long, long)} without a real GC interleaving a sample + * with a real {@code OS::nanotime()} timestamp and real heap usage, corrupting + * {@code secondsToOOM()}'s projection. Pass {@code false} to disable, {@code true} to restore. + */ + static native void setHeapFloorRecordingForTest0(boolean enabled); + + /** + * Test seam (debug native builds only): reports whether ReferenceChainTracker's search-restart + * gate ({@code canAffordNewSearch()} -> {@code hasLeakSignal()}) would currently allow a + * fresh/terminal search to start - in particular, whether {@code secondsToOOM()}'s urgent-OOM + * bypass opens this gate even with zero per-klass leak candidate (confirmable in the same test + * via {@link #selectLeakCandidateKlassIds0()}). Unlike {@link #runReferenceChainPass0()}, which + * calls {@code runPass()} unconditionally, this reads the gate itself without running a pass. + */ + static native boolean shouldRunPassForTest0(); + + /** + * Test seam (debug native builds only): the number of BFS passes run for the current/most + * recent reference-chain search. Lets a test note this count before creating an object, then + * wait for it to advance before trusting a match against that object - the only way to be + * certain the match came from a pass whose own {@code expandFrontier()} (and therefore {@code + * collectStaleExpandedEntriesForRotation()}) ran strictly after the object existed, rather + * than from the same pass racing the object's creation. + */ + static native int referenceChainPassesRunForTest0(); + // ---- Test-only reads of the current thread's OTEP record ---------------------------------- // Each resolves the current carrier's record directly (like the write primitives above) with // no cached buffer and no per-thread Java object; introspection/test use only. diff --git a/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp b/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp index 7d8ee72635..795d669941 100644 --- a/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp +++ b/ddprof-lib/src/test/cpp/livenessTracker_ut.cpp @@ -1121,6 +1121,60 @@ TEST_F(SecondsToOOMTest, RisingFloorProjectsExpectedSeconds) { EXPECT_NEAR(tracker->secondsToOOM(), 9.0, 1e-6); } +// A dip-then-recover window (usage falls for half the window, then rises +// back to where it started): the full-window fitted-line endpoints nearly +// agree (byte delta ~0), so the unguarded division would project +inf +// seconds - silently disabling the urgency ramp from this boundary forever. +// The boundary must be SKIPPED (no projection from this ring), and the +// result must be finite. Here the container ring is unavailable (no +// container limit set), so the heap ring's skip means no projection at all. +TEST_F(SecondsToOOMTest, DipThenRecoverFloorReturnsNegativeNotInf) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + // Perfectly symmetric V: pairs (i, 9-i) carry equal usage, so the + // least-squares fit's slope is exactly 0 in double arithmetic and the + // fitted endpoints agree EXACTLY (delta == 0), while the recent half + // (indices 5..9: 1500..1900) rises steeply - corroboration passes, the + // full-window denominator does not. This is precisely the shape the + // zero-denominator guard exists for; an approximately-equal pair would + // only produce a huge-but-finite projection instead of the +inf the + // guard prevents. + static const long shape[] = {1900, 1800, 1700, 1600, 1500, + 1500, 1600, 1700, 1800, 1900}; + for (int i = 0; i < 10; i++) { + tracker->heapFloorRecordForTest((u64)shape[i] * MiB, (u64)i * SEC_NS); + } + double secs = tracker->secondsToOOM(); + EXPECT_LT(secs, 0.0) + << "a dip-then-recover window must not offer a projection"; + EXPECT_TRUE(secs > -1e18 && secs < 1e18) + << "the projection must be finite, never +inf"; +} + +// Odd-length window (11 samples): pins the recent-half boundary of the +// corroboration pass (ringWindowStats' recent_min boundary is i >= n/2, so +// for odd n the median sample joins the RECENT half) and the fitted-line +// endpoint arithmetic (recent_mean at x = n-1). A rising ramp must still +// project, and the projection must match the even-window scaling of the +// same 100MiB/s rate - this is the mutation-coverage the review asked for +// on ringWindowStats' two boundary mutations (i >= n/2 and endpoint n-1). +TEST_F(SecondsToOOMTest, OddLengthRisingWindowStillProjects) { + LivenessTracker *tracker = LivenessTracker::instance(); + tracker->setMaxHeapBytesForTest((jlong)(3000 * MiB)); + // 11 samples, 100MiB apart, one second apart: 1000..2000MiB. + for (int i = 0; i < 11; i++) { + tracker->heapFloorRecordForTest(1000 * MiB + (u64)i * 100 * MiB, + (u64)i * SEC_NS); + } + double secs = tracker->secondsToOOM(); + // Rising floor, recent fitted endpoint at 2000MiB, headroom 1000MiB at + // 100MiB/s -> ~10s. Assert finite, positive, and in the right decade - + // exact value depends on the fitted endpoints the regression produces, + // which the even-window test above already pins precisely. + EXPECT_GT(secs, 0.0); + EXPECT_NEAR(secs, 9.0, 1.5); +} + // The floor's own recent-third mean has already reached the max heap size - // exhaustion is "now", not some positive number of seconds out. TEST_F(SecondsToOOMTest, FloorAtMaxHeapReturnsZero) { @@ -1190,6 +1244,40 @@ TEST_F(LeakTagPoolTest, ReleaseReturnsTagToPoolAndInfoIsInvalidated) { EXPECT_EQ(pool_size - 1, tracker->leakTagFreeCountForTest()); } +// A double release (releasing a tag that is already free) must be rejected: +// pushing the index twice would let acquireLeakTag() hand the same tag to +// two live objects. Found reachable in review via tagLeakInstances()' tag- +// adoption branch; the release path now guards on the zero/zero encoding. +TEST_F(LeakTagPoolTest, DoubleReleaseIsRejectedWithoutCorruptingTheFreeList) { + LivenessTracker *tracker = LivenessTracker::instance(); + int pool_size = tracker->leakTagPoolSizeForTest(); + + jlong tag = tracker->acquireLeakTagForTest(7, 7); + ASSERT_GE(tag, tracker->leakTagBaseForTest()); + tracker->releaseLeakTagForTest(tag); + EXPECT_EQ(pool_size, tracker->leakTagFreeCountForTest()); + + // Second release of the same tag: must be a no-op, not a second push. + tracker->releaseLeakTagForTest(tag); + EXPECT_EQ(pool_size, tracker->leakTagFreeCountForTest()) + << "double release must not grow the free list past the pool"; + + // The pool is still fully usable afterwards: draining and refilling it + // hands out exactly pool_size distinct tags, never a duplicate while + // all are live. + jlong seen[1]; + (void)seen; + int acquired = 0; + for (int i = 0; i < pool_size; i++) { + if (tracker->acquireLeakTagForTest(500 + i, 1) != 0) { + acquired++; + } + } + EXPECT_EQ(pool_size, acquired); + EXPECT_EQ(0, tracker->acquireLeakTagForTest(1, 1)) + << "pool must still exhaust at exactly LEAK_TAG_POOL_SIZE"; +} + TEST_F(LeakTagPoolTest, ReleaseOutsidePoolRangeIsIgnored) { LivenessTracker *tracker = LivenessTracker::instance(); jlong base = tracker->leakTagBaseForTest(); diff --git a/ddprof-lib/src/test/cpp/referenceChainsAnchorTests.inc b/ddprof-lib/src/test/cpp/referenceChainsAnchorTests.inc deleted file mode 100644 index af0dbdf434..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsAnchorTests.inc +++ /dev/null @@ -1,820 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -TEST_F(ReferenceChainsBfsTest, StaticAnchorRotationWalksRootAttachedStaticHolders) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - void *holderCls = (void *)0x4001, *chunkCls = (void *)0x4002; - addClass(holderCls, "Lcom/rc/descendwalk/StaticHolder;"); - int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/StaticChunk;"); - - int holderNode = addNode(); - int tableNode = addNode(); - int entryNode = addNode(); - int leakChunk = addNode(); - const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); - node_tags[leakChunk] = leak_tag; - - // Static Map -> table -> Entry -> chunk: the collection-shaped static holder's internals, - // deeper than one hop. - script = { - {JVMTI_HEAP_REFERENCE_FIELD, holderNode, tableNode, -1}, - {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, entryNode, leakChunk, chunk}, - }; - - // Seed the frontier exactly as the static sweep admits a static field's value: root-attached, - // STATIC_FIELD root kind, FRONTIER state. - FrontierTable *frontier = tracker->frontierTable(); - node_tags[holderNode] = 101; // mock_GetObjectsWithTags' resolvable tag - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 101, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - 101, 9001, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 102, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 103, 101, 1, FrontierEntryState::FRONTIER, 0)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 104, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - 104, 9004, JVMTI_HEAP_REFERENCE_JNI_GLOBAL); - - std::vector selected = - ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( - 4); - // Both DURABLE root kinds are selected (STATIC_FIELD 101, JNI_GLOBAL 104, in cursor/tag order); - // the transient-root decoy and the child are not. - ASSERT_EQ(2u, selected.size()); - EXPECT_EQ(101, selected[0]); - EXPECT_EQ(104, selected[1]); - std::vector walk_selected = {selected[0]}; - - // The static anchor's whole internal structure is admitted by one bounded walk, intercepting - // the leak tag at depth 3 below the holder. - int edges = 0; - ReferenceChainsTestAccessor::walkStaticFieldAnchorsForTest( - &mock_jvmti, &mock_jni, walk_selected, 1000, &edges); - jlong table_ftag = tags_ever_assigned[tableNode]; - jlong entry_ftag = tags_ever_assigned[entryNode]; - jlong chunk_ftag = tags_ever_assigned[leakChunk]; - ASSERT_GT(table_ftag, 0) << "table array was not reached by the anchor walk"; - ASSERT_GT(entry_ftag, 0) << "Entry was not reached one hop below table"; - ASSERT_NE(chunk_ftag, leak_tag) - << "leak-tagged chunk inside the static holder was never intercepted"; - EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); - FrontierEntry chunk_entry{}; - ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); - EXPECT_EQ(entry_ftag, chunk_entry.parent_tag); - EXPECT_EQ(3u, chunk_entry.depth); - - tracker->stop(); -} - -// Round-14 tiered selection: a container-shaped anchor (its own class implements Collection/Map - -// the LEAK_BUFFER wrapper shape) admitted at a LATE index position must leap the queue of ~28k -// other-tier anchors (the round-13 hotdog measurement: admission-order selection put the leak -// holder at position ~12-21k against ~4k walk coverage per search - deterministically unreachable). -TEST_F(ReferenceChainsBfsTest, ContainerAnchorLeapsQueueAcrossLargeIndex) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr int kAnchorCount = 28000; - for (int i = 0; i < kAnchorCount; i++) { - jlong tag = 100 + i; - jlong class_tag = 500000 + i; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - tag, class_tag, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - // Containers only at late positions 24000..24003 - buried behind 24k other-tier anchors - // under any admission-order cursor. - bool container = (i >= 24000 && i <= 24003); - ReferenceChainsTestAccessor::primeClassShapeForTest(class_tag, - container); - } - // One leak-tagged anchor at the very tail - tier 0, must lead. - jlong leak_anchor = 100 + kAnchorCount; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, leak_anchor, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - frontier->setLeakTag(leak_anchor, 777); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - leak_anchor, 599999, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - ReferenceChainsTestAccessor::primeClassShapeForTest( - 599999, false /* its tier comes from leak_tag, not shape */); - - std::vector selected = - ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( - 16); - ASSERT_EQ(16u, selected.size()) - << "28k+5 eligible anchors at a 16 budget must fill the selection"; - EXPECT_EQ(leak_anchor, selected[0]) - << "the leak-tagged anchor (tier 0) must lead the walk order"; - // The four container anchors are selected in this FIRST call despite positions 24000+ (tags - // 24100-24103). - for (jlong t : {24100, 24101, 24102, 24103}) { - EXPECT_NE(std::find(selected.begin(), selected.end(), t), selected.end()) - << "container anchor " << t - << " did not leap the other-tier queue"; - } - // Sanity: an early other-tier anchor also made the cut (cursor-fair fill from position 0). - EXPECT_NE(std::find(selected.begin(), selected.end(), (jlong)100), - selected.end()); - - tracker->stop(); -} - -// Round-15 fresh-admission priority: the hotdog wrapper is admitted LATE in a search (its holder -// class sits at sweep index 24627 of 33270, so admission lands at the anchor-index TAIL) - behind -// the whole fair container backlog measured at 1633 entries against 44-75-pass search lifetimes -// (the fair container cursor would reach it at pass ~102+, after the search is dead). -TEST_F(ReferenceChainsBfsTest, FreshContainerWalkedBeforeFairBacklog) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // A fair container backlog, "admitted long ago": 200 containers at positions 0..199. - constexpr int kBacklog = 200; - for (int i = 0; i < kBacklog; i++) { - jlong tag = 500 + i; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - tag, 700000 + i, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - ReferenceChainsTestAccessor::primeClassShapeForTest(700000 + i, true); - } - // First collector call: all 200 anchors are fresh (nothing has had a first look yet), the lane - // keeps 16 and the rest spend their first look (they fall back to the fair container tier). - std::vector first = - ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( - 16); - ASSERT_EQ(16u, first.size()) << "200 eligible containers at a 16 budget"; - - // The late admission wave: a wrapper-class container at the index tail, a fresh classified - // NON-container (a fresh String static - it must NOT ride the fresh lane), and a fresh - // unclassified anchor (a genuinely new class - it must ride the lane, so the classification lag - // cannot lose the wrapper's fresh window). - const jlong wrapper_tag = 500 + kBacklog; - const jlong fresh_string_tag = 500 + kBacklog + 1; - const jlong fresh_unknown_tag = 500 + kBacklog + 2; - for (auto [tag, class_tag, container, prime] : - {std::make_tuple(wrapper_tag, (jlong)799001, true, true), - std::make_tuple(fresh_string_tag, (jlong)799002, false, true), - std::make_tuple(fresh_unknown_tag, (jlong)799003, false, - false /* deliberately unclassified */)}) { - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - tag, class_tag, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - if (prime) { - ReferenceChainsTestAccessor::primeClassShapeForTest(class_tag, - container); - } - } - - std::vector second = - ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( - 16); - ASSERT_EQ(16u, second.size()); - EXPECT_EQ(wrapper_tag, second[0]) - << "the freshly admitted container must lead the walk order, not " - "wait behind the 184-container fair backlog"; - EXPECT_NE(std::find(second.begin(), second.end(), fresh_unknown_tag), - second.end()) - << "an unclassified fresh anchor must ride the fresh lane (the " - "wrapper admits a pass before reconcile classifies its class)"; - EXPECT_EQ(std::find(second.begin(), second.end(), fresh_string_tag), - second.end()) - << "a fresh NON-container stays in the other tier - the fresh lane " - "is the container lane, or fresh Strings would flood it"; - // The fair container lap still gets the leftover budget. Call 1's fresh picks (500-515) spent - // the whole budget, so the fair cursor never advanced - call 2's fair picks start at tag 500 - // again (a benign one-call overlap: walks are idempotent, and it only happens when fresh and - // fair coincide at a lap boundary). - EXPECT_NE(std::find(second.begin(), second.end(), (jlong)500), - second.end()) - << "the fair container lap must still advance with the leftover " - "budget"; - EXPECT_EQ(14, (int)std::count_if(second.begin(), second.end(), - [](jlong t) { - return t >= 500 && t < 500 + kBacklog; - })) - << "16 budget - 2 fresh picks = 14 fair-container picks"; - - // And the dropped fresh entries from call 1 (516-699 spent their first look) are still covered - // by the fair tier: a third call with no new admits must keep advancing the fair container - // cursor from wherever call 2 left it, not re-drain anything. - std::vector third = - ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( - 16); - ASSERT_EQ(16u, third.size()); - EXPECT_EQ(std::find(third.begin(), third.end(), wrapper_tag), - third.end()) - << "the wrapper already had its first look - it must not be " - "re-selected while the fair cursor has 184 uncovered peers"; - - tracker->stop(); -} - -// Round-14 tier fairness: the other tier (everything not leak-tagged, not container-shaped) still -// reaches full coverage across wraps - a 40-anchor tier at a 16 budget covers all 40 in exactly 3 -// calls with no duplicates within a call. -TEST_F(ReferenceChainsBfsTest, AnchorOtherTierFairCoverageAcrossWraps) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr int kOtherCount = 40; - for (int i = 0; i < kOtherCount; i++) { - jlong tag = 200 + i; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - tag, 300000 + i, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - ReferenceChainsTestAccessor::primeClassShapeForTest(300000 + i, false); - } - - std::vector all_selected; - for (int call = 0; call < 3; call++) { - std::vector selected = - ReferenceChainsTestAccessor:: - collectStaticFieldAnchorsForRotationForTest(16); - int expected = (call == 2) ? 8 : 16; - ASSERT_EQ(expected, (int)selected.size()) - << "call " << call << " should select " << expected; - std::set dedup(selected.begin(), selected.end()); - ASSERT_EQ(selected.size(), dedup.size()) - << "no anchor may be selected twice within a call"; - all_selected.insert(all_selected.end(), selected.begin(), - selected.end()); - } - std::set covered(all_selected.begin(), all_selected.end()); - ASSERT_EQ(40u, covered.size()) << "full other-tier coverage expected"; - for (int i = 0; i < kOtherCount; i++) { - EXPECT_NE(covered.find(200 + i), covered.end()) - << "other-tier anchor " << 200 + i << " never selected"; - } - - tracker->stop(); -} - -// Round-13/14 restart hygiene: discovered-instance tags are FRONTIER tags; restartSearch() resets -// the frontier table and _next_tag=1, so any surviving discovered slot either fails -// reconstructChain (observed on-pod: 'buildChainEvent failed ... -TEST_F(ReferenceChainsBfsTest, RestartSearchClearsDiscoveredInstanceTags) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // A watched candidate with one discovered instance recorded. - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 7); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(7, 4242, - false); - ASSERT_EQ(4242, ReferenceChainsTestAccessor::discoveredTagForTest(0, 0)); - ASSERT_EQ(1, ReferenceChainsTestAccessor::discoveredCountForTest(0)); - - // An anchor index entry (tag + parallel class tag) to confirm the index reset covers the - // parallel structures too. - ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( - 4242, 555, JVMTI_HEAP_REFERENCE_STATIC_FIELD); - (void)frontier; - - ReferenceChainsTestAccessor::restartSearchForTest(); - - EXPECT_EQ(0, ReferenceChainsTestAccessor::discoveredTagForTest(0, 0)) - << "stale discovered frontier tag survived restartSearch()"; - EXPECT_EQ(0, ReferenceChainsTestAccessor::discoveredCountForTest(0)); - EXPECT_TRUE(ReferenceChainsTestAccessor::anchorIndexIsEmptyForTest()) - << "anchor index (or its parallel arrays) survived restartSearch()"; - - tracker->stop(); -} - -// Round-19 (pod 289f8 — three JVMs of "canary search, 0/1 candidates found" while leak-tagged -// instances WERE intercepted and chains re-emitted): the marker->leak-tag design migration never -// updated the found criterion. -TEST_F(ReferenceChainsBfsTest, LeakTagChainMarksCanaryFound) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // Slot 0: candidate klass 7 whose discovered instance carries a leak tag (entry.leak_tag set — - // the shape a walk + tagLeakInstances correlation produces; target_tag becomes the leak tag). - ReferenceChainsTestAccessor::setCandidateCountForTest(2); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 7); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(1, 9); - // depth=1 at a durable STATIC_FIELD root (root_kind 8) so the retention filter - // (suppressChainEvent: depth==0, or depth==1 at a transient root) keeps both chains. - ASSERT_TRUE(frontier->insert(4242, 0, 7, 1, FrontierEntryState::FRONTIER, - 8, /*class_tag=*/0, -1, /*kind=*/0)); - frontier->setLeakTag(4242, 1073742079); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(7, 4242, true); - ASSERT_EQ(4242, ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); - ASSERT_TRUE(frontier->insert(5353, 0, 9, 1, FrontierEntryState::FRONTIER, - 8, 0, -1, 0)); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(9, 5353, false); - ASSERT_EQ(5353, ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(1, 0)); - - ReferenceChainsTestAccessor::buildDiscoveredInstanceChainsForTest(7, 1); - ReferenceChainsTestAccessor::buildDiscoveredInstanceChainsForTest(9, 1); - - // Slot 0 found via its leak-tag chain (bit 0 set, link recorded); slot 1's noise chain did not - // mark anything found. - EXPECT_EQ(1ULL, ReferenceChainsTestAccessor::candidateFoundBitsForTest()); - EXPECT_EQ(4242, ReferenceChainsTestAccessor::candidateFrontierTagForTest(0)); - EXPECT_EQ(0, ReferenceChainsTestAccessor::candidateFrontierTagForTest(1)) - << "noise-target chain must not mark the canary found"; - - tracker->stop(); -} - -// B' push site 1 - DEMOTION TIME (find-anchor-holder-eviction / _static_anchor_fifo): when -// improveChain() replaces a root-attached durable (STATIC_FIELD/ JNI_GLOBAL) entry with a deeper -// chain-attached path, the entry is leaving the anchor tier's eligible population at exactly that -// moment - the push must fire right there. -TEST_F(ReferenceChainsBfsTest, DemotionPushFiresWhenImproveChainEvictsRootAttachedStatic) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int parentNode = addNode(); - int holderNode = addNode(); - - // The chain edge whose delivery demotes the holder. - script = { - {JVMTI_HEAP_REFERENCE_FIELD, parentNode, holderNode, -1}, - }; - - // Seed exactly the pre-demotion shape: holder root-attached STATIC (anchor-eligible), parent a - // root-attached frontier object whose expansion delivers the deeper chain edge. - FrontierTable *frontier = tracker->frontierTable(); - node_tags[holderNode] = 105; - node_tags[parentNode] = 104; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 105, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 104, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - ReferenceChainsTestAccessor::pushPendingExpandForTest(104); - - int edges = 0; - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, - &edges); - // The chain edge was delivered: the holder's entry is now chain-attached (improveChain replaced - // the depth-0 root-attached admission), and the demotion pushed its tag into the at-risk FIFO. - FrontierEntry entry{}; - ASSERT_TRUE(frontier->lookup(105, &entry)); - EXPECT_EQ(104, entry.parent_tag); - EXPECT_EQ(1u, entry.depth); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); - - // Re-walking the same edge must NOT push twice: improveChain refuses (new depth 1 is not > - // current 1), and the set dedupes regardless. - int edges2 = 0; - ReferenceChainsTestAccessor::pushPendingExpandForTest(104); - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, - &edges2); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - - tracker->stop(); -} - -// B' push site 2 - SWEEP TIME: the static sweep's class->field edge onto an already-admitted -// CHAIN-ATTACHED entry (the admission-order eviction shape: born as a non-root child, never -// root-attached at all) must feed the FIFO. -TEST_F(ReferenceChainsBfsTest, SweepPushFiresOnStaticEdgeOntoChainAttachedHolder) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=100")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int classNode = addNode(); - int holderNode = addNode(); - // A child reachable ONLY from the holder: nothing in the scripted graph walks holderNode except - // the FIFO-drained anchor walk, so the child's admission after runPass is the end-to-end - // evidence that the sweep pushed the holder, the rotation phase drained it, and - // walkStaticFieldAnchors walked it. - int holderChildNode = addNode(); - addClass((void *)&node_tags[classNode], "Lcom/rc/statics/ChainBornHolder;"); - - script = { - // Only the sweep's static edge onto the holder, plus the holder's own child edge for the - // anchor walk to admit: nothing else reaches holderNode or holderChildNode, so the only - // possible push is the static-edge-onto-chain-attached site. - {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, holderNode, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, holderNode, holderChildNode, -1}, - }; - - // Seed the born-chain-attached shape the eviction leaves: holder already a non-root child - // (parent 104, depth 1). - FrontierTable *frontier = tracker->frontierTable(); - node_tags[holderNode] = 105; - // The child is untagged (0): the anchor walk's admission assigns it a fresh frontier tag, - // observable via node_tags after the pass. - ASSERT_EQ(0, node_tags[holderChildNode]); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 104, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - - // The rotation phase of the same pass drained the pushed tag into the anchor walk, and the walk - // admitted the holder's child - the push itself left no residue in the FIFO (drained empty) and - // never re-attributed the holder's entry (re-rooting is the documented refusal that motivated - // the FIFO in the first place). - EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - jlong childTag = tags_ever_assigned[holderChildNode]; - ASSERT_GT(childTag, 0) << "the FIFO-drained anchor walk never admitted " - "the holder's child"; - FrontierEntry childEntry{}; - ASSERT_TRUE(frontier->lookup(childTag, &childEntry)); - EXPECT_EQ(105, childEntry.parent_tag); - EXPECT_EQ(2u, childEntry.depth); - // The holder's entry is untouched by the push - the B' feed records the at-risk shape, it never - // re-attributes the entry. - FrontierEntry entry{}; - ASSERT_TRUE(frontier->lookup(105, &entry)); - EXPECT_EQ(104, entry.parent_tag); - EXPECT_EQ(1u, entry.depth); - - tracker->stop(); -} - -// B' mechanics: a chain-attached holder drained from the at-risk FIFO is descend-walked and -// intercepts a leak chunk 3 hops below it - the repair for the population the root-attached -// collector demonstrably cannot select (the negative control below). -TEST_F(ReferenceChainsBfsTest, AtRiskAnchorFifoDrainAndWalkIntercept) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - void *holderCls = (void *)0x6001, *chunkCls = (void *)0x6002; - addClass(holderCls, "Lcom/rc/descendwalk/ChainAttachedHolder;"); - int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/ChainChunk;"); - - int parentNode = addNode(); - int holderNode = addNode(); - int tableNode = addNode(); - int entryNode = addNode(); - int leakChunk = addNode(); - const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); - node_tags[leakChunk] = leak_tag; - - // Chain: parent -> holder -> table -> entry -> leak chunk. - script = { - {JVMTI_HEAP_REFERENCE_FIELD, holderNode, tableNode, -1}, - {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, entryNode, leakChunk, chunk}, - }; - - // Seed the holder exactly as the eviction leaves it: chain-attached (parent_tag = 104, depth 1, - // no root_kind). - FrontierTable *frontier = tracker->frontierTable(); - node_tags[holderNode] = 105; - // root_kind = 0: the entry is chain-attached, and a non-root entry's edge kind is not recorded - // (FrontierEntry::root_kind's own comment). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); - // The chain's root: TRANSIENT (stack local), so the collector's durable root-kind filter skips - // it too - the whole table is un-selectable. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 104, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - - // Negative control: the root-attached collector selects NOTHING from a table holding only a - // chain-attached holder and a transient root - the pre-B' behavior that stranded the - // dual-reachable population. - std::vector selected = - ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest(4); - ASSERT_TRUE(selected.empty()); - - // B': push + drain + walk reaches what the collector cannot. - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(105, 28366); - std::vector drained; - // 16 = ReferenceChainTracker::STATIC_ANCHOR_FIFO_DRAIN (private), the same per-pass drain cap - // runPassManualWalk() uses. - ASSERT_EQ(1, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 16, drained)); - ASSERT_EQ(1u, drained.size()); - EXPECT_EQ(105, drained[0].tag); - EXPECT_EQ(28366u, drained[0].klass_id); - EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - - int edges = 0; - std::vector drained_tags; - for (const auto &at_risk : drained) { - drained_tags.push_back(at_risk.tag); - } - ReferenceChainsTestAccessor::walkStaticAnchorFifoForTest( - &mock_jvmti, &mock_jni, drained_tags, 1000, &edges, nullptr); - jlong table_ftag = tags_ever_assigned[tableNode]; - jlong entry_ftag = tags_ever_assigned[entryNode]; - jlong chunk_ftag = tags_ever_assigned[leakChunk]; - ASSERT_GT(table_ftag, 0) << "table array was not reached by the anchor walk"; - ASSERT_GT(entry_ftag, 0) << "Entry was not reached one hop below table"; - ASSERT_NE(chunk_ftag, leak_tag) - << "leak-tagged chunk inside the chain-attached holder was never " - "intercepted"; - EXPECT_EQ(leak_tag, - ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); - FrontierEntry chunk_entry{}; - ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); - EXPECT_EQ(entry_ftag, chunk_entry.parent_tag); - EXPECT_EQ(4u, chunk_entry.depth); - - tracker->stop(); -} - -// B' requeue mechanics: a truncated anchor walk reports exactly the RESOLVED-but-unwalked tags, and -// requeueStaticAnchorFifoFront() restores them to the FIFO front in order with a consistent -// membership set - so an at-risk holder that lost its budget turn keeps it for the next pass -// instead of waiting for the next sweep lap. -TEST_F(ReferenceChainsBfsTest, TruncatedAnchorWalkRequeuesUnwalkedFifoTags) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - void *holderCls = (void *)0x7001; - addClass(holderCls, "Lcom/rc/descendwalk/RequeueHolder;"); - - int holderANode = addNode(); - int holderBNode = addNode(); - int tableNode = addNode(); - int entryNode = addNode(); - - // Holder A's subtree is deep enough that a budget of 2 truncates the walk after A (two edges - // admitted, budget exhausted on the descend); holder B then must come back unwalked. - script = { - {JVMTI_HEAP_REFERENCE_FIELD, holderANode, tableNode, -1}, - {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, - }; - - FrontierTable *frontier = tracker->frontierTable(); - node_tags[holderANode] = 105; - node_tags[holderBNode] = 106; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 106, 104, 1, FrontierEntryState::FRONTIER, 0)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 104, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(105, 2001); - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(106, 2002); - std::vector drained; - // 16 = STATIC_ANCHOR_FIFO_DRAIN (private), the per-pass drain cap. - ASSERT_EQ(2, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 16, drained)); - ASSERT_EQ(2u, drained.size()); - EXPECT_EQ(105, drained[0].tag); - EXPECT_EQ(106, drained[1].tag); - - int edges = 0; - std::vector drained_tags; - for (const auto &at_risk : drained) { - drained_tags.push_back(at_risk.tag); - } - std::vector unwalked; - ReferenceChainsTestAccessor::walkStaticAnchorFifoForTest( - &mock_jvmti, &mock_jni, drained_tags, 2, &edges, &unwalked); - ASSERT_EQ(1u, unwalked.size()); - EXPECT_EQ(106, unwalked[0]); - EXPECT_NE(0, tags_ever_assigned[tableNode]) - << "holder A's walk never ran - the truncation happened too early"; - - // Requeue exactly what the caller-side filter in runPassManualWalk() would requeue (here: - // everything unwalked, both FIFO-sourced), keeping the drained entries' klass so the per-class - // occupancy is restored. - std::vector requeue; - for (jlong unwalked_tag : unwalked) { - for (const auto &at_risk : drained) { - if (unwalked_tag == at_risk.tag) { - requeue.push_back(at_risk); - break; - } - } - } - ReferenceChainsTestAccessor::requeueStaticAnchorFifoFrontForTest(requeue); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(106)); - EXPECT_FALSE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); - std::vector redrained; - ASSERT_EQ(1, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 16, redrained)); - ASSERT_EQ(1u, redrained.size()); - EXPECT_EQ(106, redrained[0].tag); - - tracker->stop(); -} - -// a this-field self-edge is REAL in the heap - every java.util.Collections$Synchronized* holder -// carries mutex == this - so walking such a holder's own subtree (rotation anchor walk or BFS -// descent) re-reports the holder as its own child through that field. -TEST_F(ReferenceChainsBfsTest, - SelfEdgeFieldDoesNotDemoteRootAttachedStaticHolder) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - void *holderCls = (void *)0x7101; - int holderClsIdx = - addClass(holderCls, "Ljava/util/Collections$SynchronizedRandomAccessList;"); - int holderNode = addNode(); - - // The holder's own self-edge: referrer == referee == holder (the mutex == this field), exactly - // as its rotation walk re-reports it. - script = { - {JVMTI_HEAP_REFERENCE_FIELD, holderNode, holderNode, holderClsIdx}, - }; - - // Seed the pre-demotion shape: holder root-attached STATIC_FIELD (anchor-eligible), admitted at - // depth 0. - FrontierTable *frontier = tracker->frontierTable(); - node_tags[holderNode] = 105; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 105, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::pushPendingExpandForTest(105); - - // A delivered self-edge trips BOTH sibling guards: improveChain refuses, and the - // already-admitted block's else-if then offers the same self-parent to reparentToDurableRoot, - // which refuses too. - int edges = 0; - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, - &edges); - - // The self-edge was delivered and refused: the entry keeps its root-attached shape (the - // collector's parent_tag == 0 eligibility) and no demotion push fired. - FrontierEntry entry{}; - ASSERT_TRUE(frontier->lookup(105, &entry)); - EXPECT_EQ(0, entry.parent_tag); - EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); - EXPECT_EQ(0u, entry.depth); - EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - - // The sibling guards and the direct table calls agree: a self-parent is refused by both - // improvement paths, unconditionally. - EXPECT_FALSE(frontier->improveChain(105, 105, 0, 5, 0, -1, 0, 0)); - EXPECT_FALSE(frontier->reparentToDurableRoot(105, 105, 0, -1, 0)); - - tracker->stop(); -} - -// the B' at-risk FIFO sat cap-pinned at 1024 because three classes flooded it (klass 1: 1396 -// pushes, klass 215: 1063, klass 1733: 988+), so the LEAK_BUFFER wrapper's pushes (klass 28366) -// were dropped at the cap check and the lane designed to repair exactly its demotion never carried -// it. -TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - const u32 quota = - ReferenceChainsTestAccessor::kAtRiskPerKlassCap; - ASSERT_EQ(64u, quota); - - // The flood: only the first `quota` pushes of one class land. - for (int i = 0; i < 70; i++) { - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(1000 + i, - 1733); - } - EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - // The flood's first `quota` tags hold their slots and the excess is dropped at the quota check - // - absent from the FIFO, not queued. - EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest( - 1000 + (int)quota - 1)); - EXPECT_FALSE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest( - 1000 + (int)quota)); - - // The wrapper's push (a different class) lands despite the flood - exactly the push the pod - // dropped. - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); - EXPECT_EQ(quota + 1, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_TRUE( - ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(2000)); - - // Tag dedupe is unchanged: the same tag never enters twice. - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); - EXPECT_EQ(quota + 1, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - - // Full drain: order preserved (flood first, newcomer last), occupancy erased with the entries - - // the next flood can land again (it never owns MORE than its quota, but it is not permanently - // locked out either). - std::vector drained; - ASSERT_EQ((int)quota + 1, - ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 1024, drained)); - ASSERT_EQ(quota + 1, drained.size()); - EXPECT_EQ(1000, drained.front().tag); - EXPECT_EQ(1733u, drained.front().klass_id); - EXPECT_EQ(2000, drained.back().tag); - EXPECT_EQ(28366u, drained.back().klass_id); - for (int i = 0; i < 70; i++) { - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(1000 + i, - 1733); - } - EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - - // Partial drain at the real per-pass rate (16/pass): the flood's occupancy is 64 - 16 = 48 - // after the drain, so its next push lands (refilling its share as it drains - the flood - // self-throttles, it never starves the lane). - std::vector partial; - ASSERT_EQ(16, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 16, partial)); - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(3000, 1733); - EXPECT_EQ(quota - 16 + 1, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_TRUE( - ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(3000)); - - // Requeue restores occupancy exactly: the requeued entries occupy their slots again and drain - // in FIFO order. - std::vector requeue(partial.begin(), - partial.end()); - ReferenceChainsTestAccessor::requeueStaticAnchorFifoFrontForTest(requeue); - EXPECT_EQ(quota + 1, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - std::vector after_requeue; - ASSERT_EQ(16, - ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 16, after_requeue)); - EXPECT_EQ(1000, after_requeue.front().tag); - - // Saturated-but-diverse: 16 distinct classes at exactly their quota fill the 1024 cap, and the - // newcomer is dropped at the CAP (legitimate saturation - no eviction), not because of any - // flood. - std::vector rest; - ASSERT_EQ((int)quota - 16 + 1, - ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 1024, rest)); - for (u32 klass = 1; klass <= 16; klass++) { - for (u32 i = 0; i < quota; i++) { - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest( - 10000 + klass * 100 + i, 4000 + klass); - } - } - EXPECT_EQ(1024u, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - // The 16 saturated classes sit exactly at their per-class quota (no quota drop is even possible - // at exactly `quota` pushes), so the newcomer's absence below is the CAP's doing, not the - // quota's. - ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(20000, 28366); - EXPECT_EQ(1024u, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_FALSE( - ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(20000)); - - // Leave the FIFO drained: this test is the one that saturates it, and the reset() seam above - // now clears it for the next test regardless - but a drained ending also keeps this test - // order-independent even if that seam ever regresses again. - std::vector final_drain; - ASSERT_EQ(1024, - ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( - 1024, final_drain)); - - tracker->stop(); -} - -// Retention-edge labels (fillHopEdgeLabels()/hopLabelClassFor()): the emitted chain's per-hop -// labels must decode the JVMTI-SPECIFICATION field ordinal captured at admission - the interface -// offset, the superclass-chain order, the interface-referrer branch - and degrade to the edge KIND -// on any undecodable hop, never a fabricated name (the fail-safe contract: a wrong numbering on an -// unverified JVM degrades, it does not lie). diff --git a/ddprof-lib/src/test/cpp/referenceChainsBfsTests.inc b/ddprof-lib/src/test/cpp/referenceChainsBfsTests.inc deleted file mode 100644 index 0a0b98ea41..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsBfsTests.inc +++ /dev/null @@ -1,1275 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -namespace { - -struct ScriptedEdge { - jvmtiHeapReferenceKind kind; - int referrer_idx; // -1 = heap root (no referrer) - int referee_idx; // index into ReferenceChainsBfsTest::node_tags - int class_idx; // index into ReferenceChainsBfsTest::classes, or -1 -}; - -struct ScriptedClass { - void *klass; - const char *signature; // JVMTI class signature, e.g. "Lcom/example/Foo;" -}; - -// Retention-edge label decode fixtures (ReferenceChainsBfsTest's field_decode_hierarchy + the -// hierarchy-introspection mock slots): a fake class hierarchy the slots read, mirroring just enough -// JVMTI class shape for the spec-ordinal decoder (own-declared fields in GetClassFields order, -// direct superclass, directly implemented/extended interfaces). -struct FakeField { - void *id; // fake jfieldID - const char *name; -}; -struct FakeClass { - bool is_interface; - void *super; // fake jclass, or nullptr - std::vector interfaces; // directly implemented/extended - std::vector fields; // own-declared, GetClassFields order -}; - -} // namespace - -class ReferenceChainsBfsTest : public ::testing::Test { -protected: - jvmtiInterface_1_ jvmti_tbl{}; - _jvmtiEnv mock_jvmti{}; - JNINativeInterface_ jni_tbl{}; - JNIEnv_ mock_jni{}; - - std::unordered_map tags; - std::vector classes; - std::vector script; - std::vector node_tags; - - // node_tags[idx] mirrors "the object's *current* live JVMTI tag" (0 once releaseSearchTags() - // clears it, exactly like a real GetTag() would report after SetTag(obj, 0)). - std::vector tags_ever_assigned; - - // Tags that GetObjectsWithTags() below reports as unresolvable, simulating the referenced - // object having died (GC'd) between passes - see the resolve-or-drop tests. - std::unordered_set dead_tags; - - // When true, mock_GetObjectsWithTags() below fails outright (as if the real JVMTI call had hit - // e.g. JVMTI_ERROR_OUT_OF_MEMORY), for ReleaseSearchTagsFailureTest - simulates - // releaseSearchTags()'s own GetObjectsWithTags() call failing rather than an individual tag - // failing to resolve (dead_tags above). - bool fail_get_objects_with_tags = false; - - // When non-zero, mock_GetObjectsWithTags() below busy-waits this many nanoseconds. - u64 gotw_delay_ns = 0; - - // Synthetic frontier-holder arrays for expandFrontier()'s array-holder walk: - // mock_NewObjectArray() hands back an opaque handle, mock_SetObjectArrayElement() records its - // elements here, and mock_FollowReferences() treats every recorded element as an expansion seed - // (one hop, gated by the production callback's batch_tags) when the holder is passed as - // initial_object. - std::unordered_map> holders; - uintptr_t next_holder = 0xF00D0000; - - // FindClass(name) -> registered fake class (see mock_FindClass' own comment): names - // descendFromAnchor()'s resolutions look up ("java/lang/ClassLoader", "java/lang/ThreadGroup", - // "java/security/ProtectionDomain", "java/lang/ThreadLocal$ThreadLocalMap", - // "java/lang/Thread"). - std::unordered_map find_classes; - // Fake class returned by mock_GetObjectClass() for unregistered objects - // (walkCandidateThreadLocals()'s fresh-anchor admission path). - void *thread_class = nullptr; - - // DeleteGlobalRef call count (see mock_DeleteGlobalRef). - int global_refs_deleted_ = 0; - - jvmtiEnv *orig_jvmti = nullptr; - - static ReferenceChainsBfsTest *active_fixture; - - void SetUp() override { - active_fixture = this; - // See ReferenceChainsTestAccessor's own comment - without this, a prior test in this suite - // that drove the search to SearchState::COMPLETED/ABANDONED would make every runPass() call - // below a permanent no-op. - ReferenceChainsTestAccessor::reset(); - jvmti_tbl = jvmtiInterface_1_{}; - // start() calls VM::jvmti()->SetEventNotificationMode() - stub it and swap VM::_jvmti - // (VMTestAccessor, declared above) the same way ReferenceChainsTest's fixture does, so - // start() does not dereference the real (null, no live JVM) jvmtiEnv. - jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; - jvmti_tbl.SetTag = &mock_SetTag; - jvmti_tbl.GetTag = &mock_GetTag; - jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; - jvmti_tbl.GetClassLoader = &mock_GetClassLoader; - jvmti_tbl.GetClassSignature = &mock_GetClassSignature; - jvmti_tbl.Deallocate = &mock_Deallocate; - jvmti_tbl.FollowReferences = &mock_FollowReferences; - jvmti_tbl.IterateOverReachableObjects = &mock_IterateOverReachableObjects; - jvmti_tbl.GetObjectsWithTags = &mock_GetObjectsWithTags; - // Retention-edge label decode path (hopLabelClassFor()). - jvmti_tbl.IsInterface = &mock_IsInterface; - jvmti_tbl.GetImplementedInterfaces = &mock_GetImplementedInterfaces; - jvmti_tbl.GetClassFields = &mock_GetClassFields; - jvmti_tbl.GetFieldName = &mock_GetFieldName; - mock_jvmti.functions = &jvmti_tbl; - orig_jvmti = VMTestAccessor::getJvmti(); - VMTestAccessor::setJvmti(&mock_jvmti); - - jni_tbl = JNINativeInterface_{}; - jni_tbl.DeleteLocalRef = &mock_DeleteLocalRef; - jni_tbl.FindClass = &mock_FindClass; - jni_tbl.GetObjectClass = &mock_GetObjectClass; - jni_tbl.GetSuperclass = &mock_JniGetSuperclass; - jni_tbl.NewGlobalRef = &mock_NewGlobalRef; - jni_tbl.DeleteGlobalRef = &mock_DeleteGlobalRef; - jni_tbl.EnsureLocalCapacity = &mock_EnsureLocalCapacity; - jni_tbl.NewObjectArray = &mock_NewObjectArray; - jni_tbl.SetObjectArrayElement = &mock_SetObjectArrayElement; - jni_tbl.ExceptionCheck = &mock_ExceptionCheck; - jni_tbl.ExceptionClear = &mock_ExceptionClear; - mock_jni.functions = &jni_tbl; - } - - void TearDown() override { - VMTestAccessor::setJvmti(orig_jvmti); - active_fixture = nullptr; - } - - // Registers a fake class (matched by identity, not by any real JNI semantics) that - // resolveLoadedClasses() will discover via the mocked GetLoadedClasses(). - int addClass(void *klass, const char *signature) { - classes.push_back({klass, signature}); - return (int)classes.size() - 1; - } - - // addClass() + a mock_FindClass(name) registry entry in one step, for the classes - // descendFromAnchor()'s resolution helpers look up by name (see find_classes' own comment). - int registerClassForFindClass(void *klass, const char *name, - const char *signature) { - int idx = addClass(klass, signature); - find_classes[name] = klass; - return idx; - } - - // Adds an as-yet-untagged frontier node, returning its index into node_tags for use as a - // ScriptedEdge referrer_idx/referee_idx. - int addNode() { - node_tags.push_back(0); - tags_ever_assigned.push_back(0); - return (int)node_tags.size() - 1; - } - - // Reverse lookup from a node's synthetic identity (&node_tags[idx], see mock_FollowReferences' - // initial_object handling below) back to its index. - int indexOfNode(jobject obj) const { - for (size_t i = 0; i < node_tags.size(); i++) { - if (obj == (jobject)&node_tags[i]) { - return (int)i; - } - } - return -1; - } - - static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { - // releaseSearchTags() calls SetTag(obj, 0) on the resolved objects GetObjectsWithTags() - // (below) hands back for a frontier node - route that through node_tags[idx] directly (the - // same storage GetObjectsWithTags's resolution and the production callback's tag_ptr writes - // both key off of), so the release is actually observable, not just recorded in a side map - // nothing else reads. - int idx = active_fixture->indexOfNode(object); - if (idx >= 0) { - active_fixture->node_tags[idx] = tag; - return JVMTI_ERROR_NONE; - } - if (tag == 0) { - active_fixture->tags.erase(object); - } else { - active_fixture->tags[object] = tag; - } - return JVMTI_ERROR_NONE; - } - - static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { - int idx = active_fixture->indexOfNode(object); - if (idx >= 0) { - *tag_ptr = active_fixture->node_tags[idx]; - return JVMTI_ERROR_NONE; - } - auto it = active_fixture->tags.find(object); - *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; - return JVMTI_ERROR_NONE; - } - - static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count_ptr, - jclass **classes_ptr) { - auto &classes = active_fixture->classes; *count_ptr = (jint)classes.size(); - *classes_ptr = classes.empty() - ? nullptr - : (jclass *)malloc(sizeof(jclass) * classes.size()); - for (size_t i = 0; i < classes.size(); i++) { - (*classes_ptr)[i] = (jclass)classes[i].klass; - } - return JVMTI_ERROR_NONE; - } - - // admitStaticFieldRoots()'s app-classes-first partition (referenceChains.cpp) calls this for - // every loaded class. - static jvmtiError JNICALL mock_GetClassLoader(jvmtiEnv *, jclass, - jobject *classloader_ptr) { - *classloader_ptr = nullptr; - return JVMTI_ERROR_NONE; - } - - static jvmtiError JNICALL mock_GetClassSignature(jvmtiEnv *, jclass klass, - char **signature_ptr, - char **generic_ptr) { - for (auto &c : active_fixture->classes) { - if (c.klass == (void *)klass) { - *signature_ptr = strdup(c.signature); - if (generic_ptr != nullptr) { - *generic_ptr = nullptr; - } - return JVMTI_ERROR_NONE; - } - } - return JVMTI_ERROR_INVALID_CLASS; - } - - static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { - free(mem); - return JVMTI_ERROR_NONE; - } - - static void JNICALL mock_DeleteLocalRef(JNIEnv *, jobject) { - // no-op: this fixture's fake jobject/jclass values are not real JNI local refs. - } - - // Retention-edge label decode fixtures (fillHopEdgeLabels()/ hopLabelClassFor()): a fake class - // hierarchy the hierarchy-introspection slots below read, plus a tag -> fake jclass map (the - // decoder resolves the referrer class from its raw tag via GetObjectsWithTags - the test - // populates field_decode_classes from resolveLoadedClasses()-minted tags, and - // mock_GetObjectsWithTags consults it first). - std::unordered_map field_decode_classes; - std::unordered_map field_decode_hierarchy; - - // The hierarchy-introspection slots the decoder needs (IsInterface/ - // GetImplementedInterfaces/GetClassFields/GetFieldName on the JVMTI table, GetSuperclass on the - // JNI table - modern JVMTI dropped its own GetSuperclass). - static jvmtiError JNICALL mock_IsInterface(jvmtiEnv *, jclass cls, - jboolean *is_interface_ptr) { - auto it = active_fixture->field_decode_hierarchy.find(cls); - if (it == active_fixture->field_decode_hierarchy.end()) { - return JVMTI_ERROR_INVALID_CLASS; - } - *is_interface_ptr = it->second.is_interface ? JNI_TRUE : JNI_FALSE; - return JVMTI_ERROR_NONE; - } - static jclass JNICALL mock_JniGetSuperclass(JNIEnv *, jclass cls) { - auto it = active_fixture->field_decode_hierarchy.find(cls); - return it == active_fixture->field_decode_hierarchy.end() - ? nullptr - : (jclass)it->second.super; - } - static jvmtiError JNICALL mock_GetImplementedInterfaces( - jvmtiEnv *, jclass cls, jint *count_ptr, jclass **ifaces_ptr) { - auto it = active_fixture->field_decode_hierarchy.find(cls); - if (it == active_fixture->field_decode_hierarchy.end()) { - return JVMTI_ERROR_INVALID_CLASS; - } - const std::vector &ifaces = it->second.interfaces; - *count_ptr = (jint)ifaces.size(); - *ifaces_ptr = ifaces.empty() - ? nullptr - : (jclass *)malloc(sizeof(jclass) * ifaces.size()); - for (size_t i = 0; i < ifaces.size(); i++) { - (*ifaces_ptr)[i] = (jclass)ifaces[i]; - } - return JVMTI_ERROR_NONE; - } - static jvmtiError JNICALL mock_GetClassFields( - jvmtiEnv *, jclass cls, jint *count_ptr, jfieldID **fields_ptr) { - auto it = active_fixture->field_decode_hierarchy.find(cls); - if (it == active_fixture->field_decode_hierarchy.end()) { - return JVMTI_ERROR_INVALID_CLASS; - } - const std::vector &fields = it->second.fields; - *count_ptr = (jint)fields.size(); - *fields_ptr = fields.empty() - ? nullptr - : (jfieldID *)malloc(sizeof(jfieldID) * fields.size()); - for (size_t i = 0; i < fields.size(); i++) { - (*fields_ptr)[i] = (jfieldID)fields[i].id; - } - return JVMTI_ERROR_NONE; - } - static jvmtiError JNICALL mock_GetFieldName( - jvmtiEnv *, jclass, jfieldID field, char **name_ptr, - char ** /*signature_ptr*/, char ** /*generic_ptr*/) { - for (const auto &kv : active_fixture->field_decode_hierarchy) { - for (const FakeField &f : kv.second.fields) { - if (f.id == (void *)field) { - size_t len = strlen(f.name) + 1; - char *name = (char *)malloc(len); - memcpy(name, f.name, len); - *name_ptr = name; - return JVMTI_ERROR_NONE; - } - } - } - return JVMTI_ERROR_INVALID_FIELDID; - } - - // expandFrontier()/admitStaticFieldRoots() resolve java/lang/Object once as the holder array's - // element type - a non-null fake jclass is all it needs (the type is never introspected, only - // passed to NewObjectArray()). - static jclass JNICALL mock_FindClass(JNIEnv *, const char *name) { - auto it = active_fixture->find_classes.find(name); - if (it != active_fixture->find_classes.end()) { - return (jclass)it->second; - } - return (jclass)0xC1A55; - } - - // walkCandidateThreadLocals()'s fresh-anchor admission calls GetObjectClass(thread) - - // unregistered classes return the fixture's fake Thread class (set_thread_class) so the anchor - // entry's class tag resolves through the same mocked GetTag/tagging path. - static jclass JNICALL mock_GetObjectClass(JNIEnv *, jobject) { - return (jclass)active_fixture->thread_class; - } - - - // The production code wraps that fake jclass in a global ref (a real local ref would dangle - // across JNI-entered test seams - see _cached_object_class's own comment). - static jobject JNICALL mock_NewGlobalRef(JNIEnv *, jobject obj) { - return obj; - } - - // Counts DeleteGlobalRef calls - the deferred thread-ref teardown test - // (ThreadRefUnregisterDefersGlobalRefDeletion) asserts on the count. - static void JNICALL mock_DeleteGlobalRef(JNIEnv *, jobject) { - active_fixture->global_refs_deleted_++; - } - - static jint JNICALL mock_EnsureLocalCapacity(JNIEnv *, jint) { - return JNI_OK; - } - - // expandFrontier() calls jniExceptionCheck() after every upcall that can legally throw - // (NewObjectArray/SetObjectArrayElement/EnsureLocalCapacity failures) - this fixture's mocks - // never throw, so there is never a pending exception to report or clear. - static jboolean JNICALL mock_ExceptionCheck(JNIEnv *) { - return JNI_FALSE; - } - - static void JNICALL mock_ExceptionClear(JNIEnv *) { - // no-op: mock_ExceptionCheck() never reports a pending exception. - } - - // Hands back a fresh opaque holder handle and registers it in `holders` so - // mock_SetObjectArrayElement()/mock_FollowReferences() can find its elements. - static jobjectArray JNICALL mock_NewObjectArray(JNIEnv *, jsize, jclass, - jobject) { - jobject handle = (jobject)(active_fixture->next_holder++); - active_fixture->holders[handle] = {}; - return (jobjectArray)handle; - } - - static void JNICALL mock_SetObjectArrayElement(JNIEnv *, jobjectArray array, - jsize idx, jobject value) { active_fixture->holders[(jobject)array].push_back(value); - } - - // runPassManualWalk()'s root enumeration (the default, non-fallback path): reports each - // scripted root edge's referee to heapRootCallback() exactly as a real - // IterateOverReachableObjects() reports a root-held object - tag_ptr only, no oop, no - // transitive children (see runPassManualWalk()'s own comment). - static jvmtiError JNICALL mock_IterateOverReachableObjects( - jvmtiEnv *, jvmtiHeapRootCallback heap_root_cb, - jvmtiStackReferenceCallback, jvmtiObjectReferenceCallback, - const void *user_data) { - for (auto &e : active_fixture->script) { - if (e.referrer_idx != -1) { - continue; - } - jlong class_tag = 0; - if (e.class_idx >= 0) { - class_tag = active_fixture->tags[active_fixture->classes[e.class_idx].klass]; - } - jlong *tag_ptr = &active_fixture->node_tags[e.referee_idx]; - jvmtiIterationControl ctl = heap_root_cb( - JVMTI_HEAP_ROOT_JNI_GLOBAL, class_tag, /*size=*/0, tag_ptr, - const_cast(user_data)); - if (*tag_ptr != 0) { - active_fixture->tags_ever_assigned[e.referee_idx] = *tag_ptr; - } - if (ctl == JVMTI_ITERATION_ABORT) { - break; - } - } - return JVMTI_ERROR_NONE; - } - - // Resolves each requested tag to its node's synthetic identity (&node_tags[idx]) by scanning - // node_tags for a matching current value - mirroring real GetObjectsWithTags()'s "only - // currently-live tags come back" contract. - static jvmtiError JNICALL mock_GetObjectsWithTags( - jvmtiEnv *, jint tag_count, const jlong *req_tags, jint *count_ptr, - jobject **object_result_ptr, jlong **tag_result_ptr) { - if (active_fixture->fail_get_objects_with_tags) { - // Deliberately leave *count_ptr/*object_result_ptr/*tag_result_ptr untouched - a real - // failed JVMTI call makes no promise about them, and releaseSearchTags() must not read - // them on this path. - return JVMTI_ERROR_OUT_OF_MEMORY; - } - if (active_fixture->gotw_delay_ns != 0) { - u64 until = OS::nanotime() + active_fixture->gotw_delay_ns; - while (OS::nanotime() < until) { - // busy-wait: a sleep could overshoot by scheduler latency, and the overshoot - // direction matters for the one-batch deadline arithmetic the callers of this knob - // rely on. - } - } - std::vector objs; - std::vector found; - for (jint i = 0; i < tag_count; i++) { - jlong want = req_tags[i]; - if (want == 0 || active_fixture->dead_tags.count(want) > 0) { - continue; - } - // The decoder resolves a referrer CLASS from its raw (negative) tag - no node carries - // one, so the tag -> fake jclass map (field_decode_classes, see its own comment) serves - // it. - auto fd = active_fixture->field_decode_classes.find(want); - if (fd != active_fixture->field_decode_classes.end()) { - objs.push_back((jobject)fd->second); - found.push_back(want); - break; - } - for (size_t idx = 0; idx < active_fixture->node_tags.size(); idx++) { - if (active_fixture->node_tags[idx] == want) { - objs.push_back((jobject)&active_fixture->node_tags[idx]); - found.push_back(want); - break; - } - } - } - *count_ptr = (jint)objs.size(); - *object_result_ptr = objs.empty() - ? nullptr : (jobject *)malloc(sizeof(jobject) * objs.size()); - *tag_result_ptr = found.empty() - ? nullptr : (jlong *)malloc(sizeof(jlong) * found.size()); - for (size_t i = 0; i < objs.size(); i++) { - (*object_result_ptr)[i] = objs[i]; - (*tag_result_ptr)[i] = found[i]; - } - return JVMTI_ERROR_NONE; - } - - // Plays back `script` against the real production heap_reference_callback, modelling enough of - // FollowReferences' actual semantics for these heap-walk tests to be meaningful: - "a reference - // from A to B is not traversed until A is visited" - an edge whose referrer was not returned - // JVMTI_VISIT_OBJECTS for (or was never itself visited) is skipped, exactly as a real traversal - // would never reach it. - static jvmtiError JNICALL mock_FollowReferences( - jvmtiEnv *, jint, jclass, jobject initial_object, - const jvmtiHeapCallbacks *callbacks, const void *user_data) { - std::unordered_map expandable; // seed_idx == -2 marks the root walk (initial_object == NULL); any - // other value marks an expansion walk seeded from one or more boundary objects, in which - // case root edges are never replayed. - int seed_idx = -2; - // The transient holder array itself is never tagged (mirrors real production: - // admitStaticFieldRoots()/expandFrontier() never call SetTag on the frontier-holder array - // they build), so every holder->element ARRAY_ELEMENT edge below is replayed with a - // referrer tag of 0. - static jlong holder_tag = 0; - if (initial_object != nullptr) { - auto holder_it = active_fixture->holders.find(initial_object); - if (holder_it != active_fixture->holders.end()) { - // Array-holder walk (expandFrontier()'s already-tagged boundary batch, or - // admitStaticFieldRoots()'s negative- tagged class-object seed): actually invoke - // the production callback for each holder->element edge, exactly like a real - // FollowReferences(initial_object=holder_array) call would - this is what lets - // heap_reference_callback()'s own tag-sign/reference_kind logic (e.g. the *tag_ptr - // < 0 early-return and its admitStaticFieldRoots() carve-out) actually run, rather - // than assuming every element is expandable. - seed_idx = -1; - for (jobject elem : holder_it->second) { - int idx = active_fixture->indexOfNode(elem); - if (idx < 0) { - continue; - } - jlong *tag_ptr = &active_fixture->node_tags[idx]; - jint ctl = callbacks->heap_reference_callback( - JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, nullptr, - /*class_tag=*/0, /*referrer_class_tag=*/0, - /*size=*/0, tag_ptr, &holder_tag, - /*length=*/-1, const_cast(user_data)); - if (*tag_ptr != 0) { - active_fixture->tags_ever_assigned[idx] = *tag_ptr; - } - if (ctl & JVMTI_VISIT_ABORT) { - return JVMTI_ERROR_NONE; - } - expandable[idx] = (ctl & JVMTI_VISIT_OBJECTS) != 0; } - } else { - seed_idx = active_fixture->indexOfNode(initial_object); - expandable[seed_idx] = true; - } - } - for (auto &e : active_fixture->script) { - if (e.referrer_idx == -1) { - if (seed_idx != -2) { - continue; // resumed pass: never replay root edges - } - } else { - auto it = expandable.find(e.referrer_idx); - if (it == expandable.end() || !it->second) { continue; - } - } jlong class_tag = 0; - if (e.class_idx >= 0) { - class_tag = active_fixture->tags[active_fixture->classes[e.class_idx].klass]; - } - jlong *referrer_tag_ptr = e.referrer_idx >= 0 - ? &active_fixture->node_tags[e.referrer_idx] : nullptr; - jlong *tag_ptr = &active_fixture->node_tags[e.referee_idx]; - jint ctl = callbacks->heap_reference_callback( - e.kind, nullptr, class_tag, /*referrer_class_tag=*/0, - /*size=*/0, tag_ptr, referrer_tag_ptr, /*length=*/-1, - const_cast(user_data)); - if (*tag_ptr != 0) { - active_fixture->tags_ever_assigned[e.referee_idx] = *tag_ptr; - } - if (ctl & JVMTI_VISIT_ABORT) { - return JVMTI_ERROR_NONE; - } - expandable[e.referee_idx] = (ctl & JVMTI_VISIT_OBJECTS) != 0; - } - return JVMTI_ERROR_NONE; - } -}; - -ReferenceChainsBfsTest *ReferenceChainsBfsTest::active_fixture = nullptr; - -TEST_F(ReferenceChainsBfsTest, ReconstructsChainForSyntheticGraph) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - void *classA = (void *)0x2001, *classB = (void *)0x2002, - *classTarget = (void *)0x2003; - int ca = addClass(classA, "Lcom/rc/phase3/graph/A;"); - int cb = addClass(classB, "Lcom/rc/phase3/graph/B;"); - int ct = addClass(classTarget, "Lcom/rc/phase3/graph/Target;"); - - int nodeA = addNode(); - int nodeB = addNode(); - int nodeTarget = addNode(); - - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, ca}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, cb}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeTarget, ct}, - }; - - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_FALSE(truncated); - - // A single pass that reaches full exhaustion of the reachable graph completes the search and - // releases every tag it assigned - so the tag must be fetched via tags_ever_assigned (captured - // at assignment time), not node_tags (already reset to 0 by releaseSearchTags() by the time - // runPass() returns; see ReleasesTagsOnCompletion below for the release itself). - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - jlong targetTag = tags_ever_assigned[nodeTarget]; - ASSERT_NE(0, targetTag); - EXPECT_EQ(0, node_tags[nodeTarget]); // released - see the comment above - - std::vector chain; - ASSERT_TRUE(tracker->frontierTable()->reconstructChain(targetTag, &chain)); - ASSERT_EQ(3u, chain.size()); - - int expectedTarget = Profiler::instance()->lookupClass( - "com/rc/phase3/graph/Target", strlen("com/rc/phase3/graph/Target")); - int expectedB = Profiler::instance()->lookupClass( - "com/rc/phase3/graph/B", strlen("com/rc/phase3/graph/B")); - int expectedA = Profiler::instance()->lookupClass( - "com/rc/phase3/graph/A", strlen("com/rc/phase3/graph/A")); - ASSERT_NE(-1, expectedTarget); - ASSERT_NE(-1, expectedB); - ASSERT_NE(-1, expectedA); - - EXPECT_EQ((u32)expectedTarget, chain[0]); - EXPECT_EQ((u32)expectedB, chain[1]); - EXPECT_EQ((u32)expectedA, chain[2]); - - // buildChainEvent() wraps the same reconstructChain() call into the ReferenceChainEvent shape - // Recording::recordReferenceChain() (flightRecorder.cpp) expects - same chain/order, plus the - // target's own depth from FrontierEntry. - ReferenceChainEvent event; - ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, targetTag, - &event)); - EXPECT_EQ((u64)targetTag, event._target_tag); - EXPECT_EQ(2u, event._depth); // root(A, depth0) -> B(depth1) -> Target(depth2) - ASSERT_EQ(chain.size(), event._hops.size()); - for (size_t i = 0; i < chain.size(); i++) { - EXPECT_EQ(chain[i], event._hops[i].klass_id); - } - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, BuildChainEventFailsForUnknownTag) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - ReferenceChainEvent event; - EXPECT_FALSE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, 12345, - &event)); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, BuildAbandonedEventFailsUnlessSearchAbandoned) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // Freshly started: never abandoned (never even run a pass yet). - ReferenceChainAbandonedEvent event; - EXPECT_FALSE(tracker->buildAbandonedEvent(&event)); - - int nodeA = addNode(); - script = {{JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}}; - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - // Small graph, no caps hit -> COMPLETED, not ABANDONED. - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - EXPECT_FALSE(tracker->buildAbandonedEvent(&event)); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, HopCapStopsAdmittingBeyondCap) { - Arguments args; - // hops=1: only depth 0 (direct root references) may be admitted. - ASSERT_FALSE(args.parse("referencechains=true:hops=1:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int nodeA = addNode(); - int nodeB = addNode(); - - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // depth 0 - admitted - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // depth 1 - capped - }; - - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_FALSE(truncated); // hop cap is not truncation - a normal boundary - // Not truncated -> graph fully explored within the hop cap -> the search completes and releases - // its tags in the same call (see the previous test's comment) - fetch nodeA's tag via - // tags_ever_assigned, not node_tags. - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - - EXPECT_NE(0, tags_ever_assigned[nodeA]); - EXPECT_EQ(0, node_tags[nodeB]); // never admitted into the frontier - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, BudgetExhaustionTruncatesAndIsReported) { - Arguments args; - // budget=1: root enumeration and the expand phase draw from separate budget pools (see - // runPassManualWalk()'s own comment), each sized 1 here - root enum admits nodeA, then the - // expand phase's own 1-unit budget admits exactly one of nodeA's two children before - // exhausting. - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int nodeA = addNode(); - int nodeB = addNode(); - int nodeC = addNode(); - - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // admitted via root enum - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // admitted via expand - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeC, -1}, // expand budget exhausted - }; - - bool truncated = false; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_TRUE(truncated); - // Budget exhaustion for *this pass* leaves pending work (nodeC's own edge was never even - // attempted) - the search stays RUNNING, not COMPLETED, so no tag release happens yet and - // node_tags[nodeA]/[nodeB] are still the real assigned tags. - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - - EXPECT_NE(0, node_tags[nodeA]); - EXPECT_NE(0, node_tags[nodeB]); - EXPECT_EQ(0, node_tags[nodeC]); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, PreTaggedClassObjectsAreNeverExpandedOrAdmitted) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int classNode = addNode(); - node_tags[classNode] = -7; // simulate a class object already tagged by - // resolveLoadedClasses() before this pass - see ClassTagTable's - // tag-sign convention. - int fieldTargetNode = addNode(); - - script = { - // A root reference straight to the class object (e.g. a JVMTI_HEAP_REFERENCE_SYSTEM_CLASS - // root edge in a real walk). - {JVMTI_HEAP_REFERENCE_SYSTEM_CLASS, -1, classNode, -1}, - // A static field of that class - must never be delivered by a real FollowReferences call, - // since the class-object edge above must not return JVMTI_VISIT_OBJECTS; - // mock_FollowReferences enforces this the same way a real traversal would (see its own - // comment). - {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, fieldTargetNode, -1}, - }; - - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_FALSE(truncated); - - EXPECT_EQ(-7, node_tags[classNode]); // untouched - never treated as a - // frontier object - EXPECT_EQ(0, node_tags[fieldTargetNode]); // never reached - the class - // edge above must not expand - - tracker->stop(); -} - -// Regression test for admitStaticFieldRoots(): an object retained solely by a static field (no -// other GC root reaches it) must still be discovered. -TEST_F(ReferenceChainsBfsTest, DiscoversObjectRetainedOnlyByStaticField) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int classNode = addNode(); - int fieldTargetNode = addNode(); - addClass((void *)&node_tags[classNode], "Lcom/rc/statics/Holder;"); - - script = { - // No GC-root path to fieldTargetNode at all - it is reachable only via classNode's static - // field. - {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, fieldTargetNode, -1}, - }; - - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_FALSE(truncated); - - // classNode got tagged negative by the real resolveLoadedClasses() scan (not manually, unlike - // PreTaggedClassObjectsAreNeverExpandedOrAdmitted above), and was never itself admitted as a - // frontier object. - EXPECT_LT(node_tags[classNode], 0); - - jlong target_tag = tags_ever_assigned[fieldTargetNode]; - ASSERT_NE(0, target_tag); - - std::vector chain; - ASSERT_TRUE(tracker->frontierTable()->reconstructChain(target_tag, &chain)); - ASSERT_EQ(1u, chain.size()); - - FrontierEntry entry{}; - ASSERT_TRUE(tracker->frontierTable()->lookup(target_tag, &entry)); - EXPECT_EQ(0, entry.parent_tag); // root-attached, not a child hop - EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); - - // buildChainEvent() appends the root TYPE (the declaring class) as the chain's root-side end: - // the frontier path's terminal element is the static field's holder instance, one hop below the - // root type. - ReferenceChainEvent event; - ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, target_tag, - &event)); - int expectedHolder = Profiler::instance()->lookupClass( - "com/rc/statics/Holder", strlen("com/rc/statics/Holder")); - ASSERT_NE(-1, expectedHolder); - ASSERT_EQ(2u, event._hops.size()); - EXPECT_EQ(chain[0], event._hops[0].klass_id); // the target's own class, unchanged - EXPECT_EQ((u32)expectedHolder, event._hops[1].klass_id); - // One label per hop (recordReferenceChain() drops ALL labels when any is empty); the root-type - // hop's own edge is the unlabeled root edge (field_index -1). - ASSERT_EQ(2u, event._hops.size()); - EXPECT_FALSE(event._hops[0].edge_label.empty()); - EXPECT_FALSE(event._hops[1].edge_label.empty()); - - tracker->stop(); -} - -// Regression test for the resolveLoadedClasses() scan-skip guard: it must compare `class_count != -// _last_resolved_class_count`, not `class_count > _last_resolved_class_count`. -TEST_F(ReferenceChainsBfsTest, ResolveLoadedClassesRescansAfterClassCountShrinksAndPartiallyRegrows) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - void *classA = (void *)0x3001, *classB = (void *)0x3002, *classC = (void *)0x3003; - addClass(classA, "Lcom/rc/regress/A;"); - int idxB = addClass(classB, "Lcom/rc/regress/B;"); - - // Pass 1: both A and B loaded (count == 2) - both get resolved/tagged. - ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); - EXPECT_EQ(2, ReferenceChainsTestAccessor::lastResolvedClassCount()); - ASSERT_NE(0u, tags.count(classA)); - ASSERT_NE(0u, tags.count(classB)); - EXPECT_NE(0, tags[classA]); - EXPECT_NE(0, tags[classB]); - - // Simulate B's classloader being GC'd: GetLoadedClasses() now reports only A (count shrinks 2 - // -> 1), exactly like a real class unload. - classes.erase(classes.begin() + idxB); - ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); - EXPECT_EQ(1, ReferenceChainsTestAccessor::lastResolvedClassCount()); - - // Simulate a *different* class C loading back in, bringing the count back to 2 - the same count - // as pass 1's peak, but not the same class set. - addClass(classC, "Lcom/rc/regress/C;"); - ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); - EXPECT_EQ(2, ReferenceChainsTestAccessor::lastResolvedClassCount()); - ASSERT_NE(0u, tags.count(classC)); - EXPECT_NE(0, tags[classC]); // the regression this test guards against - - tracker->stop(); -} - -// Incremental resumption across passes (ReferenceChainTracker:: - -TEST_F(ReferenceChainsBfsTest, MultiPassResumptionReconstructsChainAcrossPasses) { - Arguments args; - // budget=1 forces each pass to admit at most one new frontier entry, so this 3-hop chain cannot - // be discovered within a single pass - exercising expandFrontier() (resumed passes), not just - // the first pass's root walk. - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int nodeA = addNode(); - int nodeB = addNode(); - int nodeTarget = addNode(); - - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeTarget, -1}, - }; - - // Drive the search to completion one pass at a time, exactly as threadLoop() would once wired - // up (each call bounded by `budget`). - bool truncated = true; - int passes_issued = 0; - while (tracker->searchState() == SearchState::RUNNING && passes_issued < 20) { - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - passes_issued++; - } - - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - EXPECT_GT(tracker->passesRun(), 1); // did not fit in a single pass - EXPECT_EQ(tracker->passesRun(), passes_issued); - - jlong targetTag = tags_ever_assigned[nodeTarget]; - ASSERT_NE(0, targetTag); - std::vector chain; - ASSERT_TRUE(tracker->frontierTable()->reconstructChain(targetTag, &chain)); - // Depth/parent_tag linkage survived resumption intact - all 3 hops walk back to a root-attached - // (depth 0) entry, which reconstructChain() requires to succeed at all (see its own "reaching - // parent_tag == 0" contract). - EXPECT_EQ(3u, chain.size()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, FrontierCapHitAbandonsImmediately) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000:framecap=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - ASSERT_EQ(1, tracker->frontierTable()->maxCapacity()); - - int nodeA = addNode(); - int nodeB = addNode(); - - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // fits (the one slot) - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // frontier cap hit - }; - - bool truncated = false; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_TRUE(truncated); - // Frontier-size cap hit -- the search abandons immediately (design doc's Termination-section - // priority 1). - ASSERT_EQ(SearchState::ABANDONED, tracker->searchState()); - ASSERT_EQ(SearchAbandonReason::FRONTIER_CAP, tracker->abandonReason()); - - // nodeA was admitted (frontier cap=1 allowed one entry), then its tag was released as part of - // this same pass's abandon handling (the mock GetObjectsWithTags() succeeds by default). - EXPECT_NE(0, tags_ever_assigned[nodeA]); - EXPECT_EQ(0, node_tags[nodeA]); - // nodeB was never admitted (frontier cap hit). - EXPECT_EQ(0, node_tags[nodeB]); - - tracker->stop(); -} - -// releaseSearchTags()'s GetObjectsWithTags() call failing must NOT be treated as "released" - see -// that method's own comment for why: marking a tag ABANDONED (or resetting _next_tag on restart) -// while its object might still be live would let a restarted search's fresh tags collide with it, -// corrupting FrontierTable's tag-uniqueness invariant. -TEST_F(ReferenceChainsBfsTest, ReleaseSearchTagsFailureBlocksTagReuseUntilItSucceeds) { - Arguments args; - // framecap=1 with a self-cycle: pass 1 admits nodeA (the frontier's only slot); the - // nodeA->nodeA edge then finds nodeA ALREADY_ADMITTED rather than hitting the frontier cap (no - // new slot is needed for an edge back to an already-tagged object), so the search stays RUNNING - // and only the no-progress detector - after NO_PROGRESS_PASS_LIMIT stale passes - can abandon - // it. - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000:framecap=1:ttl=0")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int nodeA = addNode(); - // nodeA -> nodeA self-cycle. With framecap=1, pass 1 admits nodeA; the self-cycle edge is - // ALREADY_ADMITTED, not a fresh frontier slot. - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeA, -1}, // self-cycle - }; - - long long failedBefore = - Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED); - - fail_get_objects_with_tags = true; - bool truncated = false; - // Pass 1: admits nodeA; the self-cycle keeps the pass truncated without growing the frontier - // further. - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_TRUE(truncated); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - EXPECT_NE(0, tags_ever_assigned[nodeA]); - EXPECT_NE(0, node_tags[nodeA]); - - // Run enough stale passes to trigger no-progress abandonment. - for (int i = 1; i < ReferenceChainTracker::NO_PROGRESS_PASS_LIMIT + 1; i++) { - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - } - // The next pass should abandon via no-progress. - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - ASSERT_EQ(SearchState::ABANDONED, tracker->searchState()); - - // GetObjectsWithTags() failed - nodeA's still-live tag must NOT have been cleared, and the - // failure must be counted. - EXPECT_NE(0, tags_ever_assigned[nodeA]); - EXPECT_NE(0, node_tags[nodeA]) << "tag must not be cleared when the " - "release batch itself failed"; - EXPECT_FALSE(ReferenceChainsTestAccessor::tagsReleased()); - EXPECT_EQ(failedBefore + 1, - Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)); - - // While the release is still outstanding, shouldRunPass() must force a retry unconditionally. - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); - EXPECT_EQ(SearchState::ABANDONED, tracker->searchState()) - << "must retry the release in place, not restart, while tags are " - "still unreleased"; - - // A further runPass() call retries the release; still failing. - int passesBefore = tracker->passesRun(); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(passesBefore, tracker->passesRun()); - EXPECT_NE(0, node_tags[nodeA]); - EXPECT_FALSE(ReferenceChainsTestAccessor::tagsReleased()); - EXPECT_EQ(failedBefore + 2, - Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)); - - // Once GetObjectsWithTags() starts succeeding again. - fail_get_objects_with_tags = false; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(0, node_tags[nodeA]); - EXPECT_TRUE(ReferenceChainsTestAccessor::tagsReleased()); - EXPECT_EQ(failedBefore + 2, - Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)) - << "a successful release must not itself count as a failure"; - - tracker->stop(); -} - -// Regression test for the CANARY_STUCK/frontier-wipe convergence bug: prior to this fix, -// runPass()'s canary-stuck branch fired purely off _passes_since_last_candidate_progress, so a -// search whose candidate simply had not been found yet was abandoned - and its frontier -// destructively wiped by the next restartSearch() - after only CANARY_NO_PROGRESS_PASS_LIMIT -// passes, even while the whole-graph frontier was still growing every single pass. -TEST_F(ReferenceChainsBfsTest, CanaryStuckRequiresWholeGraphFrontierAlsoStalled) { - Arguments args; - // budget=1: exactly one new frontier admission per pass, so the frontier grows every single - // pass for as long as the chain has unexplored nodes left - _passes_since_last_progress never - // leaves 0. - ASSERT_FALSE(args.parse( - "referencechains=true:hops=200:budget=1:ttl=0:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // A chain longer than the number of passes driven below, so the frontier still has pending work - // - and is still growing one node per pass - at every pass this test checks. - constexpr int kChainLength = 50; - std::vector nodes; - for (int i = 0; i < kChainLength; i++) { - nodes.push_back(addNode()); - } - script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodes[0], -1}); - for (int i = 1; i < kChainLength; i++) { - script.push_back( - {JVMTI_HEAP_REFERENCE_FIELD, nodes[i - 1], nodes[i], -1}); - } - - // A canary candidate that this graph never actually contains (no node is ever tagged with the - // candidate's marker tag) - the candidate-specific stuck counter - // (_passes_since_last_candidate_progress) climbs every pass with zero discovery progress, - // exactly like the live-pod scenario chasing a candidate deeper than the old fixed - // CANARY_NO_PROGRESS_PASS_LIMIT (30) passes could reach. - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - - bool truncated = true; - // One more pass than the old fixed CANARY_NO_PROGRESS_PASS_LIMIT: long enough that the pre-fix - // single-condition check would already have abandoned the search, but short enough that the - // 40-node chain still has unexplored work left, so the frontier is still genuinely growing - // every pass. - for (int i = 0; i < ReferenceChainTracker::CANARY_NO_PROGRESS_PASS_LIMIT + 2; - i++) { - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - ASSERT_EQ(0, tracker->passesSinceLastProgressForTest()) - << "pass " << i << ": frontier must still be growing every pass"; - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()) - << "pass " << i - << ": a canary search must not be abandoned while the " - "whole-graph frontier is still growing, even if its specific " - "candidate has not yet been found"; - } - // The narrower candidate-stuck counter climbed the whole time - this is what the old, - // single-condition check would have abandoned on alone. - EXPECT_GE(ReferenceChainsTestAccessor::passesSinceLastCandidateProgress(), - ReferenceChainTracker::CANARY_NO_PROGRESS_PASS_LIMIT); - - tracker->stop(); -} - -// Canary-lane backoff pacing (option A): a chase with unresolved candidates runs back-to-back only -// while it is fresh or making candidate progress; each pass with NO candidate progress doubles the -// spacing multiplier (_canary_backoff_mult) up to CANARY_BACKOFF_MULT_MAX, progress resets it to 1, -// and the OOM urgency ramp overrides the gate entirely. -TEST_F(ReferenceChainsBfsTest, CanaryLaneBacksOffWithoutProgressAndResetsOnProgress) { - Arguments args; - // Same shape as CanaryStuckRequiresWholeGraphFrontierAlsoStalled above: budget=1 with a long - // chain keeps the frontier growing one node per pass, so the CANARY_STUCK detector (which also - // requires a stalled frontier) never fires and the chase stays RUNNING through the whole loop - // below. - ASSERT_FALSE(args.parse( - "referencechains=true:hops=200:budget=1:ttl=0:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - constexpr int kChainLength = 50; - std::vector nodes; - for (int i = 0; i < kChainLength; i++) { - nodes.push_back(addNode()); - } - script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodes[0], -1}); - for (int i = 1; i < kChainLength; i++) { - script.push_back( - {JVMTI_HEAP_REFERENCE_FIELD, nodes[i - 1], nodes[i], -1}); - } - - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - bool truncated = true; - - // Pass 1: the candidate admission itself raises the progress mark (0 -> 1), so this counts as - // progress and the multiplier stays at 1 - back-to-back. - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(1, ReferenceChainsTestAccessor::canaryBackoffMult()); - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) - << "a fresh chase must be allowed to run back-to-back"; - - // Pass 2: no candidate progress -> first doubling (1 -> 2). - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(2, ReferenceChainsTestAccessor::canaryBackoffMult()); - ReferenceChainsTestAccessor::setCanaryBackoffForTest( - /*mult=*/2, /*ema_ms=*/100, OS::nanotime()); - EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) - << "a no-progress canary pass must hold off the next one"; - // Beyond the spacing, the chase is allowed again - the backoff paces, it never abandons. - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass( - ReferenceChainsTestAccessor::lastCanaryPassNs() + - 2ULL * 100ULL * 1000000ULL + 1)) - << "elapsed spacing must re-admit the canary pass"; - - // The OOM urgency ramp overrides the backoff gate entirely. - ReferenceChainsTestAccessor::setOomRampActive(true); - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) - << "urgency must bypass the canary backoff"; - ReferenceChainsTestAccessor::setOomRampActive(false); - - // Consecutive no-progress passes double the multiplier up to the cap (seeded at 8 so one more - // pass reaches it, the next holds it). - ReferenceChainsTestAccessor::setCanaryBackoffForTest( - /*mult=*/8, /*ema_ms=*/100, OS::nanotime()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(ReferenceChainTracker::CANARY_BACKOFF_MULT_MAX, - ReferenceChainsTestAccessor::canaryBackoffMult()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(ReferenceChainTracker::CANARY_BACKOFF_MULT_MAX, - ReferenceChainsTestAccessor::canaryBackoffMult()) - << "the multiplier must hold at its cap, not grow past it"; - - // Candidate progress (a new candidate admitted into a slot) resets the lane to back-to-back. - ReferenceChainsTestAccessor::setCandidateCountForTest(2); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(1, /*klass_id=*/987); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_EQ(1, ReferenceChainsTestAccessor::canaryBackoffMult()) - << "candidate progress must reset the spacing multiplier to 1"; - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, ChainCompletesWithoutAbandonment) { - Arguments args; - // budget=1 on a graph where each pass admits exactly one new edge until the chain is exhausted, - // then the frontier stops growing. - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:ttl=0:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int nodeA = addNode(); - int nodeB = addNode(); - int nodeC = addNode(); - int nodeD = addNode(); - - // A chain one node longer than either pass's 1-edge expand budget can fully drain in a single - // call, so each pass still ends truncated (see mock_FollowReferences()'s own comment: an - // array-holder walk chains through as many script edges as it can admit before budget aborts - // it) and there is still pending work left for the no-progress check to catch once the chain is - // fully discovered. - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeC, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeC, nodeD, -1}, - }; - - bool truncated = false; - // Pass 1: root enum admits nodeA, expand admits nodeB, aborts on nodeB->nodeC for lack of - // budget - truncated, frontier grew (progress). - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_TRUE(truncated); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - - // Pass 2: admits nodeC, aborts on nodeC->nodeD - still truncated, still growing (progress). - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_TRUE(truncated); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - - // Pass 3: admits nodeD, chain exhausted - no longer truncated, no pending frontier, natural - // completion. - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_FALSE(truncated); - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - - // Tags released. - EXPECT_NE(0, tags_ever_assigned[nodeA]); - EXPECT_EQ(0, node_tags[nodeA]); - EXPECT_NE(0, tags_ever_assigned[nodeB]); - EXPECT_EQ(0, node_tags[nodeB]); - - // No-progress (not TTL) is reported as the reason when the frontier stalls. - EXPECT_EQ(SearchAbandonReason::NONE, tracker->abandonReason()); - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, NoProgressAbandonsSearchAndReleasesTags) { - // Verify the progress-based abandonment wiring: the no-progress limit is accessible, positive, - // and reset to 0 by resetSearchStateForTest(). - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:ttl=0:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // Verify the no-progress limit is accessible and positive. - EXPECT_GT(ReferenceChainTracker::NO_PROGRESS_PASS_LIMIT, 0); - - // Verify that a fresh search starts with zero passes since last progress. - EXPECT_EQ(0, tracker->passesSinceLastProgressForTest()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, ResolveOrDropPrunesDeadFrontierEntries) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int nodeA = addNode(); - int nodeB = addNode(); - // A second root that root enumeration's own 1-unit budget (firstpassbudget=1) can't reach this - // pass - the resulting root-enum truncation makes runPassManualWalk() return before - // expandFrontier() ever runs (see its own comment on frontier-cap-hit/budget-exhausted - // root-enum truncation), so nodeA is admitted but never gets a chance to expand nodeA->nodeB. - int decoyRoot = addNode(); - - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, decoyRoot, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, - }; - - bool truncated = false; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); // pass 1 - ASSERT_TRUE(truncated); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - - jlong aTag = tags_ever_assigned[nodeA]; - ASSERT_NE(0, aTag); - // Simulate nodeA dying (collected) between pass 1 and pass 2 - GetObjectsWithTags will no - // longer report it as live. - dead_tags.insert(aTag); - - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); // pass 2: resolve-or-drop - EXPECT_FALSE(truncated); - // The dead branch was pruned for free - with nothing else pending, the search completes rather - // than staying RUNNING or being ABANDONED. - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - EXPECT_EQ(2, tracker->passesRun()); - - FrontierEntry entry{}; - ASSERT_TRUE(tracker->frontierTable()->lookup(aTag, &entry)); - EXPECT_EQ(FrontierEntryState::ABANDONED, entry.state); - - // nodeB was never discovered - nodeA's subtree was pruned, not expanded. - EXPECT_EQ(0, tags_ever_assigned[nodeB]); - - tracker->stop(); -} - -// pollWatchedTargets() (design doc's Open Question 3 bridging - -// =========================================================================== Pod-in-a-jar system -// harness (design node: design-pod-in-a-jar-harness; meta-whackamole-analysis): the REAL tracker -// loop (shouldRunPass -> runPass -> pollWatchedTargets, the exact threadLoop body) driven over the -// scripted mock heap, asserting SYSTEM INVARIANTS instead of unit symptoms. diff --git a/ddprof-lib/src/test/cpp/referenceChainsCoreTests.inc b/ddprof-lib/src/test/cpp/referenceChainsCoreTests.inc deleted file mode 100644 index ba561244d7..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsCoreTests.inc +++ /dev/null @@ -1,1288 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -TEST(RcDebugLevelTest, ParseAcceptsTrimmedSingleDigits) { - EXPECT_EQ(parseRcDebugLevel(nullptr), -1); - EXPECT_EQ(parseRcDebugLevel(""), -1); - EXPECT_EQ(parseRcDebugLevel("0"), 0); - EXPECT_EQ(parseRcDebugLevel("1"), 1); - EXPECT_EQ(parseRcDebugLevel("2"), 2); - EXPECT_EQ(parseRcDebugLevel(" 2\n"), 2); // the `echo 2 >` form - EXPECT_EQ(parseRcDebugLevel("\t1\r\n"), 1); - EXPECT_EQ(parseRcDebugLevel("3"), -1); - EXPECT_EQ(parseRcDebugLevel("12"), -1); - EXPECT_EQ(parseRcDebugLevel("-1"), -1); - EXPECT_EQ(parseRcDebugLevel("abc"), -1); - EXPECT_EQ(parseRcDebugLevel("2 garbage"), -1); -} - -TEST(RcDebugLevelTest, ReadFileParsesTrimmedAndRejectsInvalid) { - char path[] = "/tmp/rc_dbg_test_XXXXXX"; - int fd = mkstemp(path); - ASSERT_GE(fd, 0); - close(fd); - struct Case { - const char *content; - int expected; - }; - const Case cases[] = { - {"2\n", 2}, {"1", 1}, {" 2 ", 2}, {"\n1\n", 1}, - {"", -1}, {"3", -1}, {"22", -1}, {"abc", -1}, {"x", -1}, - }; - for (const Case &c : cases) { - FILE *f = fopen(path, "w"); - ASSERT_NE(f, nullptr); - EXPECT_GE(fputs(c.content, f), 0); // non-negative on success - fclose(f); - EXPECT_EQ(readRcDebugLevelFile(path), c.expected) << "content='" << c.content << "'"; - } - unlink(path); - EXPECT_EQ(readRcDebugLevelFile(path), -1); // now missing - EXPECT_EQ(readRcDebugLevelFile(nullptr), -1); -} - -TEST(RcDebugLevelTest, ReadFileRejectsSymlinkedOverride) { - // The knob path lives under world-writable /tmp: a symlink planted by a - // local user must not be followed (see readRcDebugLevelFile's lstat - // guard) - only regular files owned by root or the current user are read. - char target[] = "/tmp/rc_dbg_target_XXXXXX"; - int fd = mkstemp(target); - ASSERT_GE(fd, 0); - ASSERT_GE(write(fd, "2\n", 2), 0); - close(fd); - char link[] = "/tmp/rc_dbg_link_XXXXXX"; - int lfd = mkstemp(link); - ASSERT_GE(lfd, 0); - close(lfd); - ASSERT_EQ(unlink(link), 0); - ASSERT_EQ(symlink(target, link), 0); - EXPECT_EQ(readRcDebugLevelFile(link), -1) // symlink itself: rejected - << "symlinked override must not be followed"; - EXPECT_EQ(readRcDebugLevelFile(target), 2); // plain regular file: accepted - unlink(link); - unlink(target); -} - -TEST(RcDebugLevelTest, RefreshFollowsEnvWhenNoOverrideFile) { - // The refresh's fixed override path is machine-global; skip rather than flake on a developer - // machine that happens to have the file. - if (access("/tmp/ddprof_root/refchains_debug_level", F_OK) == 0) { - GTEST_SKIP() << "override file present on this machine"; - } - setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "1", 1); - rcDebugLevelRefresh(true); - EXPECT_EQ(rcDebugLevel(), 1); - setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1); - rcDebugLevelRefresh(true); - EXPECT_EQ(rcDebugLevel(), 2); - // invalid env value means silent, not "keep the previous level" - setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "bogus", 1); - rcDebugLevelRefresh(true); - EXPECT_EQ(rcDebugLevel(), 0); - // restore the pinned level for any later test's diagnostics - setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1); - rcDebugLevelRefresh(true); - EXPECT_EQ(rcDebugLevel(), 2); -} - -// VMTestAccessor - friend of VM (vmEntry.h), lets tests swap VM::_jvmti for a -class VMTestAccessor { -public: - static jvmtiEnv* getJvmti() { return VM::_jvmti; } - static void setJvmti(jvmtiEnv* env) { VM::_jvmti = env; } -}; - -// ReferenceChainsTestAccessor - same pattern as VMTestAccessor above, for the -class ReferenceChainsTestAccessor { -public: - static void reset() { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - delete t->_frontier; - t->_frontier = nullptr; - t->_class_tags = ClassTagTable(); - t->_last_resolved_class_count = 0; - // Without these two, a prior test's fully-swept (or partially-swept) - // admitStaticFieldRoots() state survives in this process-wide singleton and can wrongly - // skip the sweep entirely on this test's first pass if its resolved class count happens to - // match whatever an earlier test last left behind - see resetForRestart()'s identical reset - // of these same fields for the production-restart equivalent of this same contract. - t->_last_static_field_class_count = -1; - t->_static_field_sweep_cursor = 0; - t->_static_field_sweep_cycle_truncated = false; - t->_next_tag = 1; - // Shared with LivenessTracker (classTagAllocator.h) - process-wide, not - // per-ReferenceChainTracker-instance, so it needs its own reset seam rather than being a - // plain member write. - ClassTagAllocator::resetForTest(); - t->_search_started = false; - t->_tags_released = true; - t->_search_state = SearchState::RUNNING; - t->_abandon_reason = SearchAbandonReason::NONE; - t->_search_start_ns = 0; - t->_pending_expand.clear(); - t->_priority_expand.clear(); - t->_priority_expand_set.clear(); - // B' at-risk FIFO + its indexes/counters, and the round-15 fresh lane: production clears - // all of these on every restartSearch()/ resetSearchStateForTest(), but this seam predates - // the FIFO and was never given the clears - until round 16 a test left at-risk entries - // behind and only survived because LATER tests' own runPass()s drained the residue with the - // full Bfs mock. - t->_static_anchor_fifo.clear(); - t->_static_anchor_fifo_set.clear(); - t->_static_anchor_fifo_klass_counts.clear(); - t->_static_anchor_fresh_queue.clear(); - t->_last_pass_gc_finish_epoch = 0; - t->_last_pass_ns = 0; - t->_passes_run = 0; - t->_passes_since_last_progress = 0; - t->_candidate_count = 0; - t->_candidate_found_bits = 0; - memset(t->_candidate_discovered_count, 0, sizeof(t->_candidate_discovered_count)); - t->_passes_since_last_candidate_progress = 0; - t->_last_candidate_progress_mark = 0; - t->_canary_stuck_restart_count = 0; - t->_resolved_chains.clear(); - t->_safepoint_pain_budget = PainBudget(); - t->_cpu_pain_budget = PainBudget(); - t->_search_pain_ms = 0; - t->_root_kind_rotation_cursor = 1; - t->_stale_expanded_rotation_cursor = 1; - t->_static_anchor_index.clear(); - t->_static_anchor_own_class_tags.clear(); - t->_static_anchor_index_tags.clear(); - t->_anchor_container_cursor = 0; - t->_anchor_other_cursor = 0; - t->_class_shape_cache.clear(); - t->_thread_walk_anchor_cursor = 0; - memset(t->_candidate_qualifying_tid_count, 0, - sizeof(t->_candidate_qualifying_tid_count)); - t->_hop_label_cache.clear(); - t->_watched_leak_klass_count = 0; - t->_leak_signature_totals.clear(); - t->_leak_signature_prev_totals.clear(); - t->_leak_parent_fanout.clear(); - t->_borrowed_budget = 0; - t->_consecutive_under_target_passes = 0; - // Adaptive batch + lane state: NOT covered by anything above, and a prior test that drove - // expansion leaves a non-zero EMA, a live batch size, a stale pass deadline, and/or a - // mid-alternation lane toggle behind - all of which silently change the next test's - // expandFrontier() arithmetic (exact-value asserts on batch sizing only pass standalone - // otherwise). - t->_gotw_ema_call_ns = 0; - t->_gotw_batch_size = 0; - t->_pass_deadline_ns = 0; - t->_expand_lane_prefer_priority = true; - } - - // Search restart + pain budget (SearchRestartTest below) - same rationale as the pacing - // accessors above: private state a test needs to drive/observe directly. - static bool canAffordNewSearch(u64 now_ns) { - return ReferenceChainTracker::instance()->canAffordNewSearch(now_ns); - } - - static bool shouldRunPass(u64 now_ns) { - return ReferenceChainTracker::instance()->shouldRunPass(now_ns); - } - - static void setSearchPainMs(u64 ms) { - ReferenceChainTracker::instance()->_search_pain_ms = ms; - } - - static void setCandidateFrontierTagForTest(int idx, jlong tag) { - ReferenceChainTracker::instance()->setCandidateFrontierTagForTest(idx, tag); - } - static void setCandidateParentTagForTest(int idx, jlong tag) { - ReferenceChainTracker::instance()->setCandidateParentTagForTest(idx, tag); - } - static void setCandidateReferrerKlassForTest(int idx, u32 klass_id) { - ReferenceChainTracker::instance()->setCandidateReferrerKlassForTest(idx, klass_id); - } - static void setCandidateDepthForTest(int idx, u32 depth) { - ReferenceChainTracker::instance()->setCandidateDepthForTest(idx, depth); - } - - static void setCandidateCountForTest(int n) { - ReferenceChainTracker::instance()->setCandidateCountForTest(n); - } - - // Canary-lane backoff state wrappers - see _canary_backoff_mult's own comment - // (referenceChains.h). - static int canaryBackoffMult() { - return ReferenceChainTracker::instance()->canaryBackoffMultForTest(); - } - static u64 lastCanaryPassNs() { - return ReferenceChainTracker::instance()->lastCanaryPassNsForTest(); - } - static void setCanaryBackoffForTest(int mult, u64 ema_ms, u64 last_pass_ns) { - ReferenceChainTracker::instance()->setCanaryBackoffForTest(mult, ema_ms, - last_pass_ns); - } - static void setOomRampActive(bool active) { - ReferenceChainTracker::instance()->setOomRampActiveForTest(active); - } - - static int passesSinceLastCandidateProgress() { - return ReferenceChainTracker::instance()->passesSinceLastCandidateProgressForTest(); - } - - static int canaryStuckRestartCount() { - return ReferenceChainTracker::instance()->canaryStuckRestartCountForTest(); - } - - static u64 searchPainMs() { - return ReferenceChainTracker::instance()->_search_pain_ms; - } - - // Resolved-chain cache: read-only size peek and a pass-through to the private snapshot - // (drainPendingChainEvents()) and insert (cacheResolvedChain()), for ResolvedChainCacheTest - // below - same rationale as hasResolvedChainForTag()/resolvedChainCount() below. - static size_t resolvedChainCount() { - return ReferenceChainTracker::instance()->_resolved_chains.size(); - } - - static void drain(std::vector *out) { - ReferenceChainTracker::instance()->drainPendingChainEvents(out); - } - - static void cacheChain(jlong source_tag, ReferenceChainEvent event, - jlong source_tag_val, u64 source_search_ns) { - ReferenceChainTracker::instance()->cacheResolvedChain( - source_tag, std::move(event), source_tag_val, source_search_ns); - } - - static int maxResolvedChains() { - return ReferenceChainTracker::MAX_RESOLVED_CHAINS; - } - - // Target-selection bridging step: read-only peeks into the resolved-chain cache, for asserting - // exactly which klass a chain was resolved for and the tag it was reconstructed from - see - // PollWatchedTargetsTest below. - static bool hasResolvedChainForTag(jlong tag) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - return t->_resolved_chains.find(tag) != t->_resolved_chains.end(); - } - - static jlong resolvedChainSourceTag(jlong tag) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - auto it = t->_resolved_chains.find(tag); - return it == t->_resolved_chains.end() ? 0 : it->second.source_tag; - } - - // Leak-tag correlation (design C): read a frontier entry's stored leak tag, and a pass-through - // to the private buildChainEvent(), for LeakTagInterceptionTest below - same friend-accessor - // rationale as hasResolvedChainForTag() above. - static jlong frontierLeakTag(jlong tag) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - FrontierEntry entry{}; - if (t->_frontier == nullptr || !t->_frontier->lookup(tag, &entry)) { - return -1; - } - return entry.leak_tag; - } - - static void setCandidateKlassIdForTest(int idx, u32 klass_id) { - ReferenceChainTracker::instance()->setCandidateKlassIdForTest(idx, klass_id); - } - - // Round-19: the leak-tag canary found criterion (pod 289f8 — the chase was structurally - // unresolvable after the marker->leak-tag migration; see buildDiscoveredInstanceChains' own - // comment). - static u64 candidateFoundBitsForTest() { - return ReferenceChainTracker::instance()->_candidate_found_bits; - } - - static jlong candidateFrontierTagForTest(int slot) { - return ReferenceChainTracker::instance()->_candidate_frontier_tags[slot]; - } - - static void buildDiscoveredInstanceChainsForTest(u32 klass_id, - u64 current_search_ns) { - // jvmti/jni null is safe: resolveHopEdgeLabel() null-guards and degrades hop labels to kind - // labels. - ReferenceChainTracker::instance()->buildDiscoveredInstanceChains( - nullptr, nullptr, klass_id, current_search_ns); - } - - // ---- pod-in-a-jar system harness (design-pod-in-a-jar-harness) ---- - static u8 searchStateForTest() { - return ReferenceChainTracker::instance()->_search_state; - } - - static int sweepGateResolvedCountForTest() { - return ReferenceChainTracker::instance()->_last_resolved_class_count; - } - - static int sweepGateStaticCountForTest() { - return ReferenceChainTracker::instance()->_last_static_field_class_count; - } - - static int sweepCursorForTest() { - return ReferenceChainTracker::instance()->_static_field_sweep_cursor; - } - - static int passesRunForTest() { - return ReferenceChainTracker::instance()->_passes_run; - } - - static int candidateCountForTest() { - return ReferenceChainTracker::instance()->_candidate_count; - } - - static u32 candidateKlassIdForTest(int slot) { - return ReferenceChainTracker::instance()->_candidate_klass_ids[slot]; - } - - static size_t resolvedChainCountForTest() { - return ReferenceChainTracker::instance()->_resolved_chains.size(); - } - - static std::vector resolvedChainTargetsForTest() { - std::vector out; - for (auto &kv : - ReferenceChainTracker::instance()->_resolved_chains) { - out.push_back(kv.second.event._target_tag); - } - return out; - } - - static size_t staticAnchorFreshQueueSizeForTest() { - return ReferenceChainTracker::instance() - ->_static_anchor_fresh_queue.size(); - } - - static jlong candidateDiscoveredTagForTest(int slot, int idx) { - return ReferenceChainTracker::instance()->candidateDiscoveredTagForTest(slot, idx); - } - - static int candidateDiscoveredCountForTest(int slot) { - return ReferenceChainTracker::instance()->candidateDiscoveredCountForTest(slot); - } - - // recordDiscoveredInstance()/correlateAdmittedLeakTag() are the production paths for the - // leak-correlation tests below. - static void recordDiscoveredInstanceForTest(u32 klass_id, jlong tag, - bool leak_correlated) { - ReferenceChainTracker::instance()->recordDiscoveredInstance(klass_id, tag, - leak_correlated); - } - - // Drive restartSearch() directly (the accessor base already set _tags_released, so its assert - // is satisfied). - static void restartSearchForTest() { - ReferenceChainTracker::instance()->restartSearch(); - } - - static bool anchorIndexIsEmptyForTest() { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - return t->_static_anchor_index.empty() && - t->_static_anchor_own_class_tags.empty() && - t->_static_anchor_index_tags.empty(); - } - - // Read back discovered-instance slots (frontier tags recorded by recordDiscoveredInstance). - static jlong discoveredTagForTest(int slot, int idx) { - return ReferenceChainTracker::instance() - ->_candidate_discovered_tags[slot][idx]; - } - - static int discoveredCountForTest(int slot) { - return ReferenceChainTracker::instance() - ->_candidate_discovered_count[slot]; - } - - static size_t priorityExpandCap() { - return ReferenceChainTracker::PRIORITY_EXPAND_CAP; - } - - static int maxDiscoveredPerClass() { - return ReferenceChainTracker::MAX_DISCOVERED_INSTANCES_PER_CLASS; - } - - static bool buildChainEventForTest(jvmtiEnv *jvmti, JNIEnv *jni, - jlong tag, ReferenceChainEvent *out) { - return ReferenceChainTracker::instance()->buildChainEvent(jvmti, jni, - tag, out); - } - - // Direct expandFrontier() drive for the AIMD batch test: a full runPass() drains a small graph - // to completion and its rotation phase adds extra GetObjectsWithTags calls, so per-call AIMD - // assertions cannot be made deterministic through runPass(). - static void pushPendingExpandForTest(jlong tag) { - ReferenceChainTracker::instance()->_pending_expand.push_back(tag); - } - - static void expandFrontierForTest(jvmtiEnv *jvmti, JNIEnv *jni, - int *edges_admitted) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - bool truncated = false; - bool cap_hit = false; - u64 safepoint_ticks = 0; - t->expandFrontier(jvmti, jni, t->_hop_cap, 1000, edges_admitted, - &truncated, &cap_hit, &safepoint_ticks); - } - - // Pause-time pacing controller: read-only peeks at the controller's derived values, and a - // pass-through to the private updatePacing() itself, for ReferenceChainsPacingTest below - same - // rationale as hasResolvedChainForTag()/resolvedChainCount() above (the target-selection - // bridging step): private state a test needs to drive/ observe directly, exposed via this - // existing friend accessor rather than adding public getters/setters to ReferenceChainTracker - // itself. - static int effectiveBudget() { - return ReferenceChainTracker::instance()->_effective_budget; - } - - static u64 effectiveCadenceNs() { - return ReferenceChainTracker::instance()->_effective_cadence_ns; - } - - static void updatePacing(u64 pass_wall_ns) { - ReferenceChainTracker::instance()->updatePacing(pass_wall_ns); - } - - static u64 baselineCadenceNs() { return ReferenceChainTracker::PASS_CADENCE_NS; } - - // Test-only seams for PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling below, which needs - // to start from a controlled below-ceiling/above- baseline point with a freshly reset - // controller (see that test's own comment for why chaining directly off a prior constant-input - // sequence would leave _pause_pid's integral state mid-recovery from that sequence's windup, - // muddying this method's per-step direction assertions with a transient the test is not about). - static void setEffectiveBudget(int v) { - ReferenceChainTracker::instance()->_effective_budget = v; - } - - static void setEffectiveCadenceNs(u64 v) { - ReferenceChainTracker::instance()->_effective_cadence_ns = v; - } - - static void resetPacingController() { - ReferenceChainTracker::instance()->_pause_pid.reset(); - } - - // Budget-borrowing (referenceChains.h's _borrowed_budget comment): the configured multiplier - // PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling below asserts convergence against, - // instead of hardcoding it a second time in the test itself. - static int borrowCeilingMultiplier() { - return ReferenceChainTracker::BORROW_CEILING_MULTIPLIER; - } - - static int64_t borrowedBudget() { - return ReferenceChainTracker::instance()->_borrowed_budget; - } - - // MaybeRevokeBorrowForRootEnumPass* tests below: drive the borrow state directly into "already - // granted" before exercising the revocation-only seam, and a pass-through to that seam itself - - // same rationale as updatePacing()'s own accessor above. - static void setBorrowedBudget(int64_t v) { - ReferenceChainTracker::instance()->_borrowed_budget = v; - } - - static int consecutiveUnderTargetPasses() { - return ReferenceChainTracker::instance()->_consecutive_under_target_passes; - } - - static void setConsecutiveUnderTargetPasses(int v) { - ReferenceChainTracker::instance()->_consecutive_under_target_passes = v; - } - - static void maybeRevokeBorrowForRootEnumPass(u64 pass_wall_ticks) { - ReferenceChainTracker::instance()->maybeRevokeBorrowForRootEnumPass( - pass_wall_ticks); - } - - // ReleaseSearchTagsFailureTest below: read-only peek at whether the tracker still owes a tag - // release before it can allow a restart - see _tags_released's own comment. - static bool tagsReleased() { - return ReferenceChainTracker::instance()->_tags_released; - } - - // ResolveLoadedClassesRescansAfterClassCountShrinksAndPartiallyRegrows below: direct - // pass-through to the private resolveLoadedClasses(), plus a read-only peek at the count it - // stashes - the same rationale as tagsReleased() above (private state/behavior a test needs to - // drive/observe directly, without going through a full runPass()/search lifecycle that - // resolveLoadedClasses() alone does not need). - static void resolveLoadedClasses(jvmtiEnv *jvmti, JNIEnv *jni) { - ReferenceChainTracker::instance()->resolveLoadedClasses(jvmti, jni); - } - - static int lastResolvedClassCount() { - return ReferenceChainTracker::instance()->_last_resolved_class_count; - } - - // Durability re-verification test seams: direct pass-throughs to the private tie-break/rotation - // methods, plus FrontierTable::insert() itself (also private-by-convention here in the sense - // that production code only ever calls it via admitObject()) so tests can set up a frontier - // entry's exact starting root_kind/state/parent_tag without needing a live JVMTI mock for - // IterateOverReachableObjects/FollowReferences (neither is mocked in this file - see the file - // header's FollowReferences- only mock rationale). - // Direct seam for buildCanaryChainEvent() - private in production (only - // pollWatchedTargets() calls it), but the bounded parent-chain walk and its - // cycle-corruption behavior are unit-testable only through it. - static bool buildCanaryChainEventForTest(int candidate_idx, - ReferenceChainEvent *out) { - return ReferenceChainTracker::instance()->buildCanaryChainEvent( - candidate_idx, out); - } - - static bool insertFrontierEntry(FrontierTable *frontier, jlong tag, - jlong parent_tag, u32 depth, u8 state, - u8 root_kind, u32 referrer_klass = 0, - jlong class_tag = 0, - jint referrer_field_index = -1, - u8 edge_kind = 0, - jlong referrer_class_tag = 0) { - return frontier->insert(tag, parent_tag, referrer_klass, depth, - state, root_kind, class_tag, - referrer_field_index, edge_kind, - referrer_class_tag); - } - - static bool maybeUpgradeRootAttachedRootKind(FrontierTable *frontier, - jlong tag, - u8 new_root_kind) { - return ReferenceChainTracker::instance() - ->maybeUpgradeRootAttachedRootKind(frontier, tag, new_root_kind); - } - - static std::vector collectStaleRootKindEntriesForRotation( - int max_count) { - return ReferenceChainTracker::instance() - ->collectStaleRootKindEntriesForRotation(max_count); - } - - static std::vector collectStaleExpandedEntriesForRotation( - int max_count) { - return ReferenceChainTracker::instance() - ->collectStaleExpandedEntriesForRotation(max_count); - } - - // Candidate-scoped reach (descendFromAnchor()/walkCandidateThreadLocals()/ - // walkStaticFieldAnchors()): direct drives for the same reason as expandFrontierForTest() above - // - a full runPass() drains a small graph to completion and its other phases add interference, - // so the walk phases are exercised on their own. - static void walkCandidateThreadLocalsForTest(jvmtiEnv *jvmti, JNIEnv *jni, - int budget, - int *edges_admitted) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - bool truncated = false; - bool cap_hit = false; - u64 safepoint_ticks = 0; - t->walkCandidateThreadLocals(jvmti, jni, budget, edges_admitted, - &truncated, &cap_hit, &safepoint_ticks); - } - - static void walkStaticFieldAnchorsForTest(jvmtiEnv *jvmti, JNIEnv *jni, - const std::vector &tags, - int budget, - int *edges_admitted) { - walkStaticAnchorFifoForTest(jvmti, jni, tags, budget, edges_admitted, - nullptr); - } - - static std::vector - collectStaticFieldAnchorsForRotationForTest(int max_count) { - return ReferenceChainTracker::instance() - ->collectStaticFieldAnchorsForRotation(max_count); - } - - static void addToStaticAnchorIndexForTest(jlong tag, jlong own_class_tag, - u8 root_kind) { - ReferenceChainTracker::instance() - ->addToStaticAnchorIndex(tag, own_class_tag, root_kind); - } - - // Prime the class-shape cache as if reconcileAnchorClassShapes() had classified `class_tag` - // (tests script shapes instead of driving the JNI interface walk, which needs real classes). - static void primeClassShapeForTest(jlong class_tag, bool container) { - ReferenceChainTracker::instance()->_class_shape_cache[class_tag] = - container - ? (u8)ReferenceChainTracker::AnchorClassShape::CONTAINER - : (u8)ReferenceChainTracker::AnchorClassShape::NON_CONTAINER; - } - - // B' at-risk static-anchor FIFO (see _static_anchor_fifo's declaration comment in - // referenceChains.h). - using AtRiskAnchor = ReferenceChainTracker::AtRiskAnchor; - static constexpr u32 kAtRiskPerKlassCap = - ReferenceChainTracker::STATIC_ANCHOR_ATRISK_PER_KLASS_CAP; - - static void pushStaticAnchorFifoForTest(jlong tag, u32 klass_id) { - ReferenceChainTracker::instance()->pushAtRiskStaticAnchor(tag, - klass_id); - } - - static int drainStaticAnchorFifoForTest( - int max_count, - std::vector &out) { - return ReferenceChainTracker::instance()->drainStaticAnchorFifo( - max_count, out); - } - - static void requeueStaticAnchorFifoFrontForTest( - const std::vector &entries) { - ReferenceChainTracker::instance()->requeueStaticAnchorFifoFront( - entries); - } - - static size_t staticAnchorFifoSizeForTest() { - return ReferenceChainTracker::instance()->_static_anchor_fifo.size(); - } - - static bool staticAnchorFifoContainsForTest(jlong tag) { - return ReferenceChainTracker::instance() - ->_static_anchor_fifo_set.contains(tag); - } - - static void walkStaticAnchorFifoForTest(jvmtiEnv *jvmti, JNIEnv *jni, - const std::vector &tags, - int budget, int *edges_admitted, - std::vector *unwalked) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - bool truncated = false; - bool cap_hit = false; - u64 safepoint_ticks = 0; - t->walkStaticFieldAnchors(jvmti, jni, tags, budget, edges_admitted, - &truncated, &cap_hit, &safepoint_ticks, - unwalked); - } - - // Direct candidate-slot seeding (the production path fills these via pollWatchedTargets()'s - // snapshot loop - see _candidate_qualifying_tids' own comment): the walk phase tests need - // exactly one (slot, klass, tid) combination without driving LivenessTracker's hysteresis - // machinery. - static void seedCandidateSlotForTest(int slot, u32 klass_id, - const jint *tids, int tid_count) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - t->_candidate_klass_ids[slot] = klass_id; - for (int i = 0; i < tid_count; i++) { - t->_candidate_qualifying_tids[slot][i] = tids[i]; - } - t->_candidate_qualifying_tid_count[slot] = tid_count; - if (slot + 1 > t->_candidate_count) { - t->_candidate_count = slot + 1; - } - } - - static int candidateQualifyingTidCountForTest(int slot) { - return ReferenceChainTracker::instance() - ->_candidate_qualifying_tid_count[slot]; - } - - static jlong getTagForTest(jvmtiEnv *jvmti, jobject obj) { - return ReferenceChainTracker::instance()->getTag(jvmti, obj); - } - - // Snapshot of _priority_expand's current contents, in queue order - used by tests to check for - // duplicate tags after both rotation collectors have run against it within the same simulated - // pass. - static std::vector priorityExpandContents() { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - return std::vector(t->_priority_expand.begin(), - t->_priority_expand.end()); - } - - // StaleExpandedRotationSkipsPreexistingQueueEntries below: simulates a tag left in - // _priority_expand by a prior pass's truncated expandFrontier() batch (expandFrontier()'s own - // "leave the batch at the front of the source queue for a later pass to retry" comment) without - // driving a full expandFrontier()/JVMTI round-trip to produce one. - static void pushPriorityExpand(jlong tag) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - t->_priority_expand.push_back(tag); - t->_priority_expand_set.insert(tag); - } - - // Simulates expandFrontier() having fully drained _priority_expand at the end of a pass (the - // common case: rotation's whole selection fit within that pass's rotation_budget slice) - see - // StaleExpandedRotationStarvesHighTagEntryBehindLowTagPopulation below, which needs this to - // model collectStaleExpandedEntriesForRotation() being called fresh on each of several - // simulated passes, the way runPassManualWalk() actually does it once per real pass. - static void clearPriorityExpand() { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - t->_priority_expand.clear(); - t->_priority_expand_set.clear(); - } - - static void setRootKindRotationCursor(jlong tag) { - ReferenceChainTracker::instance()->_root_kind_rotation_cursor = tag; - } - - static jlong rootKindRotationCursor() { - return ReferenceChainTracker::instance()->_root_kind_rotation_cursor; - } - - static int rootKindRotationBudget() { - return ReferenceChainTracker::ROOT_KIND_ROTATION_BUDGET; - } - - static int staleExpandedRotationBudget() { - return ReferenceChainTracker::STALE_EXPANDED_ROTATION_BUDGET; - } - - static size_t priorityExpandSize() { - return ReferenceChainTracker::instance()->_priority_expand.size(); - } - - // Snapshot of _pending_expand's current contents, in queue order - used by the rolling-resume - // smoke test to verify that a truncated expandFrontier() batch pops fully-processed entries - // (mark EXPANDED) and leaves only the partially-processed and unvisited entries at the front of - // the queue for the next pass to retry. - static std::vector pendingExpandContents() { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - return std::vector(t->_pending_expand.begin(), - t->_pending_expand.end()); - } - - static size_t pendingExpandSize() { - return ReferenceChainTracker::instance()->_pending_expand.size(); - } - - // Self-calibrating adaptive batch size (AIMD): read/write the per-call EMA and live batch size - // so tests can verify the AIMD dynamics. - static u64 gotwEmaCallNs() { - return ReferenceChainTracker::instance()->_gotw_ema_call_ns; - } - - static void setGotwEmaCallNs(u64 v) { - ReferenceChainTracker::instance()->_gotw_ema_call_ns = v; - } - - static size_t gotwBatchSize() { - return ReferenceChainTracker::instance()->_gotw_batch_size; - } - - static void setGotwBatchSize(size_t v) { - ReferenceChainTracker::instance()->_gotw_batch_size = v; - } - - // Read-only peeks at the batch-control constants (private statics - friendship applies inside - // this class's methods, not in test bodies). - static u64 gotwCpuBudgetNs() { - return ReferenceChainTracker::GOTW_CPU_BUDGET_NS; - } - - static size_t gotwInitialBatchSize() { - return (size_t)ReferenceChainTracker::GOTW_INITIAL_BATCH_SIZE; - } - - static size_t gotwMinBatch() { - return ReferenceChainTracker::GOTW_MIN_BATCH; - } - - static size_t gotwMaxBatch() { - return ReferenceChainTracker::GOTW_MAX_BATCH; - } - - static size_t gotwBacklogMinDepth() { - return ReferenceChainTracker::GOTW_BACKLOG_MIN_DEPTH; - } - - static u64 gotwBacklogWindowMult() { - return ReferenceChainTracker::GOTW_BACKLOG_WINDOW_MULT; - } - - // gotwWindowNs() is a pure function of (remaining window, lane depth) and the seeded EMA - - // directly unit-testable without a mock JVMTI call. - static u64 gotwWindowNs(u64 remaining_ns, size_t lane_depth) { - return ReferenceChainTracker::instance()->gotwWindowNs(remaining_ns, - lane_depth); - } - - static void setPassDeadlineNs(u64 v) { - ReferenceChainTracker::instance()->_pass_deadline_ns = v; - } - - static bool expandLanePreferPriority() { - return ReferenceChainTracker::instance()->_expand_lane_prefer_priority; - } - - // Leak-tag pool range base (private static) - same friend-access rationale as the AIMD - // constants above. - static jlong leakTagBase() { - return ReferenceChainTracker::LEAK_TAG_BASE; - } - - // Leak-accumulation rotation test seams (collectLeakAccumulationCandidatesForRotation()). - static void setWatchedLeakKlassIdsForTest(const std::vector &ids) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - int n = (int)std::min(ids.size(), - (size_t)ReferenceChainTracker::MAX_WATCHED_LEAK_KLASSES); - for (int i = 0; i < n; i++) { - t->_watched_leak_klass_ids[i] = ids[i]; - } - t->_watched_leak_klass_count = n; - } - - static void trackLeakAccumulation(FrontierTable *frontier, u32 referrer_klass, - jlong parent_tag, jlong tag) { - ReferenceChainTracker::instance()->trackLeakAccumulation( - frontier, referrer_klass, parent_tag, tag); - } - - static std::vector collectLeakAccumulationCandidatesForRotation( - int max_count) { - return ReferenceChainTracker::instance() - ->collectLeakAccumulationCandidatesForRotation(max_count); - } - - static int leakAccumulationRotationBudget() { - return ReferenceChainTracker::LEAK_ACCUMULATION_ROTATION_BUDGET; - } - - static u32 leakSignatureTotal(u32 leaf_klass_id, u32 parent_class_id) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - u64 key = t->leakSignatureKey(leaf_klass_id, parent_class_id); - auto it = t->_leak_signature_totals.find(key); - return it != t->_leak_signature_totals.end() ? it->second : 0; - } - - static u32 leakParentFanout(jlong parent_tag) { - ReferenceChainTracker *t = ReferenceChainTracker::instance(); - auto it = t->_leak_parent_fanout.find(parent_tag); - return it != t->_leak_parent_fanout.end() ? it->second.fanout : 0; - } - - static size_t leakSignatureCount() { - return ReferenceChainTracker::instance()->_leak_signature_totals.size(); - } - - static void seedLeakAccumulationForNewlyWatchedKlass(u32 klass_id) { - ReferenceChainTracker::instance() - ->seedLeakAccumulationForNewlyWatchedKlass(klass_id); - } -}; - -static jvmtiError JNICALL mock_SetEventNotificationMode(jvmtiEnv *, jvmtiEventMode, - jvmtiEvent, jthread, ...) { - return JVMTI_ERROR_NONE; -} - -class ReferenceChainsTest : public ::testing::Test { -protected: - jvmtiInterface_1_ tbl{}; - _jvmtiEnv mock_env{}; - jvmtiEnv *orig_jvmti = nullptr; - - void SetUp() override { - orig_jvmti = VMTestAccessor::getJvmti(); - tbl = jvmtiInterface_1_{}; - tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; - mock_env.functions = &tbl; - VMTestAccessor::setJvmti(&mock_env); - } - - void TearDown() override { - VMTestAccessor::setJvmti(orig_jvmti); - } -}; - -TEST_F(ReferenceChainsTest, DefaultDisabled) { - Arguments args; - EXPECT_FALSE(args._reference_chains); -} - -TEST_F(ReferenceChainsTest, FlagParsesEnabled) { - Arguments args; - Error error = args.parse("referencechains=true"); - EXPECT_FALSE(error); - EXPECT_TRUE(args._reference_chains); -} - -TEST_F(ReferenceChainsTest, FlagParsesDisabled) { - Arguments args; - Error error = args.parse("referencechains=false"); - EXPECT_FALSE(error); - EXPECT_FALSE(args._reference_chains); -} - -TEST_F(ReferenceChainsTest, FlagParsesSubOptions) { - Arguments args; - Error error = args.parse("referencechains=true:hops=64:budget=2000:ttl=5000:framecap=128"); - EXPECT_FALSE(error); - EXPECT_TRUE(args._reference_chains); - EXPECT_EQ(64, args._reference_chains_hop_cap); - EXPECT_EQ(2000, args._reference_chains_budget); - EXPECT_EQ(5000, args._reference_chains_ttl_ms); - EXPECT_EQ(128, args._reference_chains_frontier_cap); -} - -// Negative/out-of-range sub-options must be floored/clamped at the parse boundary -// (Arguments::parse(), arguments.cpp) rather than stored verbatim - see that call site's own -// comment for why an unclamped negative hops in particular is dangerous: `depth >= -// (u32)ctx->hop_cap` (referenceChains.cpp) casts a negative int to u32, wrapping to ~4e9 and -// silently disabling the hop cap entirely. -TEST_F(ReferenceChainsTest, FlagClampsNegativeSubOptions) { - Arguments args; - Error error = args.parse( - "referencechains=true:hops=-1:budget=-5:ttl=-1:framecap=-3:" - "pausetarget=-1:painbudget=-10"); - EXPECT_FALSE(error); - EXPECT_TRUE(args._reference_chains); - // Floored to a sane minimum (1), not left negative - a negative value cast to u32 downstream - // would otherwise wrap to a huge positive number. - EXPECT_GT(args._reference_chains_hop_cap, 0); - EXPECT_GT(args._reference_chains_budget, 0); - EXPECT_GT(args._reference_chains_frontier_cap, 0); - // ttl/pausetarget are floored at 0 (their own downstream gates already treat 0 as "disabled", - // so 0 - not 1 - is the correct floor). - EXPECT_GE(args._reference_chains_ttl_ms, 0); - EXPECT_GE(args._reference_chains_pause_target_ms, 0); - // painbudget is a percentage - clamped into [0, 100]. - EXPECT_GE(args._reference_chains_pain_budget_percent, 0); - EXPECT_LE(args._reference_chains_pain_budget_percent, 100); -} - -// A too-large painbudget must be clamped down to 100, not stored verbatim - the sibling of -// FlagClampsNegativeSubOptions above, for the upper bound rather than the lower one. -TEST_F(ReferenceChainsTest, FlagClampsOversizedPainBudgetPercent) { - Arguments args; - Error error = args.parse("referencechains=true:painbudget=250"); - EXPECT_FALSE(error); - EXPECT_EQ(100, args._reference_chains_pain_budget_percent); -} - -TEST_F(ReferenceChainsTest, FlagWithOtherArgsDoesNotClobberOuterParse) { - Arguments args; - Error error = args.parse("event=cpu,referencechains=true:hops=32,interval=1000000"); - EXPECT_FALSE(error); - EXPECT_TRUE(args._reference_chains); - EXPECT_EQ(32, args._reference_chains_hop_cap); - EXPECT_STREQ("cpu", args._event); - EXPECT_EQ(1000000, args._interval); -} - -TEST_F(ReferenceChainsTest, StartStopDisabledDoesNotCrash) { - Arguments args; - Error error = args.parse("referencechains=false"); - ASSERT_FALSE(error); - - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - Error startError = tracker->start(args); - EXPECT_FALSE(startError); - EXPECT_FALSE(tracker->enabled()); - tracker->stop(); -} - -TEST_F(ReferenceChainsTest, StartStopEnabledDoesNotCrash) { - Arguments args; - Error error = args.parse("referencechains=true:hops=10:budget=100"); - ASSERT_FALSE(error); - - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - Error startError = tracker->start(args); - EXPECT_FALSE(startError); - EXPECT_TRUE(tracker->enabled()); - tracker->stop(); -} - -// GC signal (GarbageCollectionStart/Finish -> epoch counters). - -TEST_F(ReferenceChainsTest, GCCallbacksIncrementEpochWhenEnabled) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - u64 startBefore = tracker->gcStartEpoch(); - u64 finishBefore = tracker->gcFinishEpoch(); - - ReferenceChainTracker::GarbageCollectionStart(nullptr); - ReferenceChainTracker::GarbageCollectionFinish(nullptr); - - EXPECT_EQ(startBefore + 1, tracker->gcStartEpoch()); - EXPECT_EQ(finishBefore + 1, tracker->gcFinishEpoch()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsTest, GCCallbacksAreNoOpWhenDisabled) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=false")); - - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - ASSERT_FALSE(tracker->enabled()); - - u64 startBefore = tracker->gcStartEpoch(); - u64 finishBefore = tracker->gcFinishEpoch(); - - ReferenceChainTracker::GarbageCollectionStart(nullptr); - ReferenceChainTracker::GarbageCollectionFinish(nullptr); - - EXPECT_EQ(startBefore, tracker->gcStartEpoch()); - EXPECT_EQ(finishBefore, tracker->gcFinishEpoch()); -} - -// Tag round-trip (SetTag/GetTag/clear). - -class ReferenceChainsTagTest : public ::testing::Test { -protected: - jvmtiInterface_1_ tbl{}; - _jvmtiEnv mock_env{}; - std::unordered_map tags; - - static ReferenceChainsTagTest *active_fixture; - - void SetUp() override { - active_fixture = this; - tbl = jvmtiInterface_1_{}; - tbl.SetTag = &mock_SetTag; - tbl.GetTag = &mock_GetTag; - mock_env.functions = &tbl; - } - - void TearDown() override { - active_fixture = nullptr; - } - - static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { - if (tag == 0) { - active_fixture->tags.erase(object); - } else { - active_fixture->tags[object] = tag; - } - return JVMTI_ERROR_NONE; - } - - static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { - auto it = active_fixture->tags.find(object); - *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; - return JVMTI_ERROR_NONE; - } -}; - -ReferenceChainsTagTest *ReferenceChainsTagTest::active_fixture = nullptr; - -TEST_F(ReferenceChainsTagTest, TagRoundTripsThenClears) { - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - - jlong tag = tracker->tagObject(&mock_env, obj); - EXPECT_NE(0, tag); - EXPECT_EQ(tag, tracker->getTag(&mock_env, obj)); - - tracker->clearTag(&mock_env, obj); - EXPECT_EQ(0, tracker->getTag(&mock_env, obj)); -} - -TEST_F(ReferenceChainsTagTest, TagsAreUniqueAndNeverZero) { - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - int a = 0, b = 0; - jlong tagA = tracker->tagObject(&mock_env, reinterpret_cast(&a)); - jlong tagB = tracker->tagObject(&mock_env, reinterpret_cast(&b)); - - EXPECT_NE(0, tagA); - EXPECT_NE(0, tagB); - EXPECT_NE(tagA, tagB); -} - -TEST_F(ReferenceChainsTagTest, UntaggedObjectReadsBackZero) { - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - int untagged = 0; - EXPECT_EQ(0, tracker->getTag(&mock_env, reinterpret_cast(&untagged))); -} - -// FrontierTable (tag-indexed frontier metadata table). - -TEST(FrontierTableTest, InsertThenLookupRoundTrips) { - FrontierTable table(64); - - ASSERT_TRUE(table.insert(1, /*parent_tag=*/0, /*referrer_klass=*/7, - /*depth=*/0, FrontierEntryState::FRONTIER)); - - FrontierEntry entry{}; - ASSERT_TRUE(table.lookup(1, &entry)); - EXPECT_EQ(0, entry.parent_tag); - EXPECT_EQ(7u, entry.referrer_klass); - EXPECT_EQ(0u, entry.depth); - EXPECT_EQ(FrontierEntryState::FRONTIER, entry.state); -} - -TEST(FrontierTableTest, LookupOfNeverInsertedTagFails) { - FrontierTable table(64); - FrontierEntry entry{}; - EXPECT_FALSE(table.lookup(1, &entry)); - EXPECT_FALSE(table.lookup(5, &entry)); -} - -TEST(FrontierTableTest, NonPositiveTagIsRejected) { - FrontierTable table(64); - FrontierEntry entry{}; - EXPECT_FALSE(table.insert(0, 0, 0, 0)); - EXPECT_FALSE(table.insert(-1, 0, 0, 0)); - EXPECT_FALSE(table.lookup(0, &entry)); - EXPECT_FALSE(table.lookup(-1, &entry)); -} - -TEST(FrontierTableTest, LookupLockedRejectsNonPositiveTag) { - FrontierTable table(64); - ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); - FrontierEntry entry{}; - // tag=0 must be rejected the same way lookup() rejects it - callers index the table with tag-1, - // so a `tag <= 0` check (not just `tag < 0`) is required to keep that subtraction from wrapping - // into a valid slot. - EXPECT_FALSE(table.lookupLocked(0, &entry)); - EXPECT_FALSE(table.lookupLocked(-1, &entry)); -} - -TEST(FrontierTableTest, LookupLockedRejectsTagPastCurrentSize) { - FrontierTable table(64); - ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); - FrontierEntry entry{}; - // Only tag=1 has ever been inserted (table_size == 1); tag=2 maps to idx=1, exactly at the - // current size boundary, and must be rejected rather than read out of bounds. - EXPECT_FALSE(table.lookupLocked(2, &entry)); -} - -TEST(FrontierTableTest, ParentTagChainReconstructsAcrossHops) { - // Mirrors how the heap-walk engine walks parent_tag links back to a root: insert a small chain - // root(tag=1) <- mid(tag=2) <- leaf(tag=3) and confirm the links resolve in order. - FrontierTable table(64); - ASSERT_TRUE(table.insert(1, 0, 100, 0, FrontierEntryState::EDGE)); - ASSERT_TRUE(table.insert(2, 1, 200, 1, FrontierEntryState::EDGE)); - ASSERT_TRUE(table.insert(3, 2, 300, 2, FrontierEntryState::EDGE)); - - FrontierEntry entry{}; - jlong tag = 3; - std::vector chain; - while (tag != 0) { - ASSERT_TRUE(table.lookup(tag, &entry)); - chain.push_back(entry.referrer_klass); - tag = entry.parent_tag; - } - - ASSERT_EQ(3u, chain.size()); - EXPECT_EQ(300u, chain[0]); - EXPECT_EQ(200u, chain[1]); - EXPECT_EQ(100u, chain[2]); -} - -TEST(FrontierTableTest, ClearMarksAbandonedWithoutRemovingEntry) { - FrontierTable table(64); - ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); - - table.clear(1); - - FrontierEntry entry{}; - ASSERT_TRUE(table.lookup(1, &entry)); - EXPECT_EQ(FrontierEntryState::ABANDONED, entry.state); -} - -TEST(FrontierTableTest, ClearOfNeverInsertedTagIsNoOp) { - FrontierTable table(64); - table.clear(1); // must not crash - FrontierEntry entry{}; - EXPECT_FALSE(table.lookup(1, &entry)); -} - -TEST(FrontierTableTest, GrowsPastInitialCapacityUpToMaxCap) { - // Force at least one resize by inserting beyond the small max_cap. - const int max_cap = 10; - FrontierTable table(max_cap); - ASSERT_LE(table.capacity(), max_cap); - - for (jlong tag = 1; tag <= max_cap; tag++) { - ASSERT_TRUE(table.insert(tag, tag - 1, (u32)tag, (u32)(tag - 1))) - << "insert failed for tag " << tag; - } - EXPECT_EQ(max_cap, table.capacity()); - - for (jlong tag = 1; tag <= max_cap; tag++) { - FrontierEntry entry{}; - ASSERT_TRUE(table.lookup(tag, &entry)); - EXPECT_EQ((u32)tag, entry.referrer_klass); - } -} - -TEST(FrontierTableTest, CapacityExhaustedReportsFailureInsteadOfCrashing) { - const int max_cap = 4; - FrontierTable table(max_cap); - - for (jlong tag = 1; tag <= max_cap; tag++) { - ASSERT_TRUE(table.insert(tag, 0, 0, 0)); - } - // One past max_cap must be rejected, not silently dropped-but-crashing. - EXPECT_FALSE(table.insert(max_cap + 1, 0, 0, 0)); - EXPECT_EQ(max_cap, table.capacity()); - - // Existing entries remain intact after the failed insert. - FrontierEntry entry{}; - EXPECT_TRUE(table.lookup(1, &entry)); -} - -TEST(FrontierTableTest, ZeroMaxCapRejectsEveryInsert) { - FrontierTable table(0); - EXPECT_EQ(0, table.capacity()); - EXPECT_FALSE(table.insert(1, 0, 0, 0)); -} - -TEST(FrontierTableTest, ConcurrentInsertWhileGrowingDoesNotCrash) { - // Small max_cap relative to thread/tag count forces repeated resizes while other threads are - // concurrently inserting distinct tags. - const int max_cap = 4096; - const int thread_count = 8; - const int tags_per_thread = 256; - FrontierTable table(max_cap); - - std::vector threads; - for (int t = 0; t < thread_count; t++) { - threads.emplace_back([&table, t, tags_per_thread]() { - for (int i = 0; i < tags_per_thread; i++) { - jlong tag = (jlong)t * tags_per_thread + i + 1; - table.insert(tag, 0, (u32)tag, 0); - } - }); - } - for (auto &th : threads) { - th.join(); - } - - int found = 0; - for (jlong tag = 1; tag <= (jlong)thread_count * tags_per_thread; tag++) { - FrontierEntry entry{}; - if (table.lookup(tag, &entry)) { - EXPECT_EQ((u32)tag, entry.referrer_klass); - found++; - } - } - // Every tag fits well within max_cap, so all inserts must have succeeded and be independently - // readable. - EXPECT_EQ(thread_count * tags_per_thread, found); -} - -TEST(FrontierTableTest, ReconstructChainWalksParentTagsAndPreservesState) { - FrontierTable table(64); - ASSERT_TRUE(table.insert(1, 0, 100, 0, FrontierEntryState::FRONTIER)); - ASSERT_TRUE(table.insert(2, 1, 200, 1, FrontierEntryState::FRONTIER)); - ASSERT_TRUE(table.insert(3, 2, 300, 2, FrontierEntryState::EXPANDED)); - - std::vector chain; - ASSERT_TRUE(table.reconstructChain(3, &chain)); - ASSERT_EQ(3u, chain.size()); - EXPECT_EQ(300u, chain[0]); - EXPECT_EQ(200u, chain[1]); - EXPECT_EQ(100u, chain[2]); - - // Every visited entry must KEEP its pre-walk state: the earlier EDGE demotion made - // resolved-path holders invisible to the rotation collectors (which select EXPANDED entries), - // so later leak instances behind a changed holder were never re-discovered. - for (jlong tag = 1; tag <= 2; tag++) { - FrontierEntry entry{}; - ASSERT_TRUE(table.lookup(tag, &entry)); - EXPECT_EQ(FrontierEntryState::FRONTIER, entry.state); - } - FrontierEntry entry{}; - ASSERT_TRUE(table.lookup(3, &entry)); - EXPECT_EQ(FrontierEntryState::EXPANDED, entry.state); -} - -TEST(FrontierTableTest, ReconstructChainOfNeverInsertedTagFails) { - FrontierTable table(64); - std::vector chain; - EXPECT_FALSE(table.reconstructChain(1, &chain)); -} - -// Heap-walk engine (ReferenceChainTracker::runPass()/ - diff --git a/ddprof-lib/src/test/cpp/referenceChainsEventTests.inc b/ddprof-lib/src/test/cpp/referenceChainsEventTests.inc deleted file mode 100644 index 314c3b4ece..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsEventTests.inc +++ /dev/null @@ -1,455 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // Hierarchy mirroring the spec's own numbering example shape: interface IBase { int p; } 1 own - // field interface ISub extends IBase { int x; } 1 own field class Base { int base_f; } 1 own - // field, no super class Holder extends Base implements ISub { int holder_a; Object leakList; } - // interface ISink { Object CONST_A; Object CONST_B; } Spec ordinal spaces - // (jvmtiHeapReferenceInfoField): Holder (class branch): base = ISub(1) + IBase(1) = 2 - // (transitive interfaces, each once); then the superclass chain root-first: base_f@2; then own - // fields in GetClassFields order: holder_a@3, leakList@4. - void *ibase = (void *)0x5001, *isub = (void *)0x5002, *base = (void *)0x5003, - *holder = (void *)0x5004, *isink = (void *)0x5005; - addClass(ibase, "Lcom/rc/labels/IBase;"); - addClass(isub, "Lcom/rc/labels/ISub;"); - addClass(base, "Lcom/rc/labels/Base;"); - addClass(holder, "Lcom/rc/labels/Holder;"); - addClass(isink, "Lcom/rc/labels/ISink;"); - ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); - // resolveLoadedClasses minted each class's raw (negative) tag via the mock's tag map - read - // them back for the decoder's tag->class lookup and the entries' referrer_class_tag values. - auto tagOf = [&](void *k) -> jlong { return tags[k]; }; - field_decode_classes[tagOf(holder)] = holder; - field_decode_classes[tagOf(isink)] = isink; - field_decode_classes[tagOf(base)] = base; - // ibase deliberately NOT registered into field_decode_classes: an unresolvable referrer class - // below must degrade to a kind label. - field_decode_hierarchy[ibase] = {true, nullptr, {}, {{(void *)0x6001, "p"}}}; - field_decode_hierarchy[isub] = - {true, nullptr, {ibase}, {{(void *)0x6002, "x"}}}; - field_decode_hierarchy[base] = {false, nullptr, {}, {{(void *)0x6003, "base_f"}}}; - field_decode_hierarchy[holder] = - {false, base, {isub}, - {{(void *)0x6004, "holder_a"}, {(void *)0x6005, "leakList"}}}; - field_decode_hierarchy[isink] = - {true, nullptr, {}, - {{(void *)0x6006, "CONST_A"}, {(void *)0x6007, "CONST_B"}}}; - - FrontierTable *frontier = tracker->frontierTable(); - // Chain: [chunk(3)] <- Base.base_f(ordinal 0 over Base's space) <- [value2(2), class Base] <- - // Holder.leakList(ordinal 4, the static root edge with the declaring class as referrer) <- - // [static value(1), class Holder] <- [class Holder (the ROOT TYPE - buildChainEvent() appends - // the root-attached entry's referrer_class_tag for static-field roots)]. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EDGE, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, - /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), - /*referrer_field_index=*/4, /*edge_kind=*/0, - /*referrer_class_tag=*/tagOf(holder))); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 2, 1, 1, FrontierEntryState::EDGE, /*root_kind=*/0, - /*referrer_klass=*/0, /*class_tag=*/tagOf(base), - /*referrer_field_index=*/3, JVMTI_HEAP_REFERENCE_FIELD)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 3, 2, 2, FrontierEntryState::EDGE, /*root_kind=*/0, - /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), - /*referrer_field_index=*/0, JVMTI_HEAP_REFERENCE_FIELD)); - - ReferenceChainEvent event; - ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( - &mock_jvmti, &mock_jni, /*target_tag=*/3, &event)); - ASSERT_EQ(4u, event._hops.size()); - // Leaf first: chunk is retained via Base.base_f (parent entry's class is Base, ordinal 0 in - // Base's own space), then value2 via Holder's holder_a (ordinal 3 = interface offset 2 + Base's - // 1 + own position 0), then the static root edge's field name leakList (ordinal 4), then the - // root-type hop (class Holder) - the root edge itself, kind label only. - EXPECT_EQ("base_f", event._hops[0].edge_label); - EXPECT_EQ("holder_a", event._hops[1].edge_label); - EXPECT_EQ("leakList", event._hops[2].edge_label); - EXPECT_EQ("static_field", event._hops[3].edge_label); - int expectedRootType = Profiler::instance()->lookupClass( - "com/rc/labels/Holder", strlen("com/rc/labels/Holder")); - ASSERT_NE(-1, expectedRootType); - EXPECT_EQ((u32)expectedRootType, event._hops[3].klass_id); - - // Interface-referrer branch: ISink's own-field ordinals have NO superclass-chain component - // (base = superinterfaces' fields only). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 4, 0, 0, FrontierEntryState::EDGE, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, - /*referrer_klass=*/0, /*class_tag=*/tagOf(isink), - /*referrer_field_index=*/1, /*edge_kind=*/0, - /*referrer_class_tag=*/tagOf(isink))); - ReferenceChainEvent iface_event; - ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( - &mock_jvmti, &mock_jni, /*target_tag=*/4, &iface_event)); - // Same root-type append as above: the root-side end gains ISink (the declaring class of the - // static field) plus its kind-only root edge. - ASSERT_EQ(2u, iface_event._hops.size()); - EXPECT_EQ("CONST_B", iface_event._hops[0].edge_label); - EXPECT_EQ("static_field", iface_event._hops[1].edge_label); - int expectedSinkRoot = Profiler::instance()->lookupClass( - "com/rc/labels/ISink", strlen("com/rc/labels/ISink")); - ASSERT_NE(-1, expectedSinkRoot); - EXPECT_EQ((u32)expectedSinkRoot, iface_event._hops[1].klass_id); - - // Fail-safe: a referrer class that cannot be resolved degrades to the edge KIND label, never a - // fabricated name. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 5, 0, 0, FrontierEntryState::EDGE, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, - /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), - /*referrer_field_index=*/0, /*edge_kind=*/0, - /*referrer_class_tag=*/tagOf(ibase))); - ReferenceChainEvent degraded_event; - ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( - &mock_jvmti, &mock_jni, /*target_tag=*/5, °raded_event)); - // Both hops degrade to kind labels: the holder hop's referrer class (IBase) is deliberately - // unregistered from the decoder, and the appended root-type hop (IBase itself) carries no field - // identity. - ASSERT_EQ(2u, degraded_event._hops.size()); - EXPECT_EQ("static_field", degraded_event._hops[0].edge_label); - EXPECT_EQ("static_field", degraded_event._hops[1].edge_label); - int expectedIbaseRoot = Profiler::instance()->lookupClass( - "com/rc/labels/IBase", strlen("com/rc/labels/IBase")); - ASSERT_NE(-1, expectedIbaseRoot); - EXPECT_EQ((u32)expectedIbaseRoot, degraded_event._hops[1].klass_id); - - tracker->stop(); -} - -// PRIORITY_EXPAND_CAP backpressure: with the fast lane at the cap, the rotation collectors must -// stop pushing. -TEST_F(ReferenceChainsBfsTest, PriorityExpandCapStopsRotationCollectorPushes) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // One eligible stale-EXPANDED entry. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - - for (size_t i = 0; i < ReferenceChainsTestAccessor::priorityExpandCap(); i++) { - ReferenceChainsTestAccessor::pushPriorityExpand((jlong)(100 + i)); - } - std::vector selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(10); - EXPECT_TRUE(selected.empty()) - << "collector must stop pushing once _priority_expand hits the cap"; - - // With the lane drained (a pass's expand phase consumed it), the collector selects again. - ReferenceChainsTestAccessor::clearPriorityExpand(); - selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(10); - ASSERT_EQ(1u, selected.size()); - EXPECT_EQ((jlong)1, selected[0]); - - tracker->stop(); -} - -// The stale-expansion rotation must select leak parents from _leak_parent_fanout ahead of the blind -// table lap: the fanout entries are the EXPANDED parents that actually lead to watched leak-klass -// children, and neither the blind lap (~table_size/budget passes, hundreds live) nor the -// growth-gated leak-accumulation tier reaches them in steady state. -TEST_F(ReferenceChainsBfsTest, StaleRotationPrefersLeakParentsOverBlindLap) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - // Fanout parent 1 and an unrelated stale-EXPANDED entry 3. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, - /*class_tag=*/42)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 3, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 1, 10); - - // Budget 1: the fanout parent wins over the blind-lap entry. - std::vector selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(1); - ASSERT_EQ(1u, selected.size()); - EXPECT_EQ((jlong)1, selected[0]); - - // Budget covering both: fanout parent first, blind lap fills the rest. - ReferenceChainsTestAccessor::clearPriorityExpand(); - selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(2); - ASSERT_EQ(2u, selected.size()); - EXPECT_EQ((jlong)1, selected[0]); - EXPECT_EQ((jlong)3, selected[1]); - - tracker->stop(); -} - -// reparentToDurableRoot: a depth-1 entry first admitted through a transient root (stack/JNI local) -// is re-parented to a durable root-attached parent at equal depth - the case improveChain() cannot -// express (it requires a strictly deeper path), and exactly the hotdog shape where the singleton -// collection is a depth-0 static root and its elements depth 1. -TEST_F(ReferenceChainsBfsTest, ReparentToDurableRootSwapsTransientForDurable) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // tag 1: transient root (old parent). tag 2: target at depth 1 under it. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_JNI_LOCAL)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 2, 1, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - // tag 5: durable static root (new parent). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 5, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - // tag 6: another transient root - must never be swapped TO. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 6, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - - EXPECT_TRUE(frontier->reparentToDurableRoot(2, 5, 42)); - FrontierEntry entry{}; - ASSERT_TRUE(frontier->lookup(2, &entry)); - EXPECT_EQ((jlong)5, entry.parent_tag); - EXPECT_EQ((u32)42, entry.referrer_klass); - - // Transient new parent: no swap (would trade one noise root for another). - EXPECT_FALSE(frontier->reparentToDurableRoot(2, 6, 43)); - ASSERT_TRUE(frontier->lookup(2, &entry)); - EXPECT_EQ((jlong)5, entry.parent_tag) << "parent must be unchanged"; - - // Depth-2 targets are out of scope (judging root durability there would require walking both - // chains). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 7, 2, 2, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - EXPECT_FALSE(frontier->reparentToDurableRoot(7, 5, 44)); - - tracker->stop(); -} - -// recordDiscoveredInstance eviction: noise instances fill discovery slots first-come-first-served, -// but a leak-correlated discovery must evict a noise slot when all are full - without eviction, the -// 8 noise instances observed on-pod permanently blocked every later leak-tagged instance of the -// watched class. -TEST_F(ReferenceChainsBfsTest, RecordDiscoveredInstanceEvictsNoiseSlots) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kKlass = 3; - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, kKlass); - - const int cap = ReferenceChainsTestAccessor::maxDiscoveredPerClass(); - for (int d = 0; d < cap; d++) { - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( - kKlass, /*tag=*/100 + d, /*leak_correlated=*/false); - } - EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - - // Noise beyond the cap is dropped, slots unchanged. - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( - kKlass, 108, false); - EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - EXPECT_EQ((jlong)100, - ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); - - // Leak-correlated discovery evicts the first noise slot (tag 100 has no frontier entry -> - // treated as uncorrelated). - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( - kKlass, 200, true); - EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - EXPECT_EQ((jlong)200, - ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); - EXPECT_EQ((jlong)101, - ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 1)); - - // Once every slot is leak-correlated, a further leak discovery is dropped (no eviction of real - // signal). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 200, 0, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - frontier->setLeakTag(200, ReferenceChainsTestAccessor::leakTagBase() + 1); - // Entries for the remaining noise slots so the eviction scan finds all slots leak-tagged. - for (int d = 1; d < cap; d++) { - jlong tag = 101 + (d - 1); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - frontier->setLeakTag(tag, ReferenceChainsTestAccessor::leakTagBase() + 2); - } - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( - kKlass, 201, true); - EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - for (int d = 0; d < cap; d++) { - EXPECT_NE((jlong)201, - ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, d)); - } - - tracker->stop(); -} - -// correlateAdmittedLeakTag: a tracked instance the BFS admitted BEFORE it was leak-tagged carries a -// frontier tag on the object; correlating stores the leak tag ON the entry (chain events then emit -// targetTag = leak tag) and records the instance as discovered. -TEST_F(ReferenceChainsBfsTest, CorrelateAdmittedLeakTagSetsEntryAndDiscovers) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kKlass = 3; - constexpr jlong kLeakTag = 0x40000000LL + 5; - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, kKlass); - - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 300, 0, 1, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - - EXPECT_TRUE(tracker->correlateAdmittedLeakTag(300, kLeakTag, kKlass)); - EXPECT_EQ(kLeakTag, (jlong)ReferenceChainsTestAccessor::frontierLeakTag(300)); - EXPECT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - EXPECT_EQ((jlong)300, - ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); - - // Idempotent: an already-correlated entry just returns true. - EXPECT_TRUE(tracker->correlateAdmittedLeakTag(300, kLeakTag, kKlass)); - EXPECT_EQ(kLeakTag, (jlong)ReferenceChainsTestAccessor::frontierLeakTag(300)); - - // Unknown tag: no crash, no discovery side effects. - EXPECT_FALSE(tracker->correlateAdmittedLeakTag(999, kLeakTag, kKlass)); - EXPECT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - - tracker->stop(); -} - -// Retention-explanation gate on the discovered-instance chains (depth==0 always suppressed; -// depth==1 suppressed only for TRANSIENT roots - a depth-1 chain from a durable root is the real -// direct-retention shape): transient depth-1 must NOT be cached, durable depth-1 and depth-2 must. -TEST_F(PollWatchedTargetsTest, DiscoveredChainGateSuppressesTransientDepthOne) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*klass_id=*/3, /*rep=*/(jweak)obj); - - // First poll populates the candidate slots from LivenessTracker's population. - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - FrontierTable *frontier = tracker->frontierTable(); - // Noise shape: transient root (JNI local frame) -> depth-1 instance. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 6, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_JNI_LOCAL)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 7, 6, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - // Real direct-retention shape: static-field root -> depth-1 instance. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 8, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 9, 8, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - // Deeper chain through the transient root: passes on depth alone. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 10, 7, 2, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - // Depth-0 transient root: the candidate instance itself held by a live frame - suppressed like - // the depth-1 transient shape. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 11, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - // Depth-0 durable root: the candidate instance IS the static field's value (the - // singleton-collection-itself shape) - a real direct-retention chain, NOT suppressible as - // noise. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 12, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 7, false); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 9, false); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 10, false); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 11, false); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 12, false); - ASSERT_EQ(5, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)) - << "depth-1 chain rooted at a transient (JNI local) root is noise"; - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(9)) - << "depth-1 chain rooted at a durable (static field) root is a real " - "direct-retention chain"; - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(10)) - << "depth-2 chain passes the gate regardless of root kind"; - EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag(11)) - << "depth-0 chain rooted at a transient (stack local) root is noise"; - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(12)) - << "depth-0 chain rooted at a durable (static field) root is the " - "direct-retention shape the search exists to report"; - - tracker->stop(); -} - -// Orphan slot sweep: a candidate that qualified long enough for the walk to record discovered -// instances, then stopped qualifying (its trend aged out of the poll's candidate list), must still -// get chains built for those instances. -TEST_F(PollWatchedTargetsTest, OrphanedSlotBuildsDiscoveredChainsAfterCandidateDropsOut) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*klass_id=*/3, /*rep=*/(jweak)obj); - - // First poll admits klass 3 into candidate slot 0 (nothing discovered yet, so nothing is built - // here). - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - ASSERT_EQ(0, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - - // The walk discovered an instance while the candidate still qualified: the real - // direct-retention shape (static-field root -> depth-1 instance), which the discovered-chain - // gate lets through. - FrontierTable *frontier = tracker->frontierTable(); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 9, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 8, 9, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); - ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 8, false); - ASSERT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); - - // The candidate stops qualifying: LivenessTracker's population table is wiped, so - // selectLeakCandidates() returns 0 on every poll from here on. - LivenessTracker::instance()->klassPopulationResetForTest(); - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(8)) - << "discovered instances recorded while the candidate qualified must " - "still get chains built after it stops qualifying"; - - tracker->stop(); -} diff --git a/ddprof-lib/src/test/cpp/referenceChainsPodTests.inc b/ddprof-lib/src/test/cpp/referenceChainsPodTests.inc deleted file mode 100644 index bc4bbafa8b..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsPodTests.inc +++ /dev/null @@ -1,479 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -class PodInAJarTest : public ReferenceChainsBfsTest { -protected: - struct PodTopology { - int holder_class_node = -1; - int wrapper_node = -1; - int list_node = -1; - int leak_cls_idx = -1; - int wrapper_cls_idx = -1; - int list_cls_idx = -1; - std::vector chunk_nodes; - std::vector chunk_leak_tags; - std::vector flood_class_nodes; - std::vector flood_nodes; - std::vector filler_nodes; - }; - - static constexpr int kChunks = 6; // < MAX_DISCOVERED_INSTANCES_PER_CLASS - static constexpr u64 kCycleNs = 2000000000ULL; // 2s fake-clock step - - void SetUp() override { - ReferenceChainsBfsTest::SetUp(); - // resolveCandidateRepresentative() NewLocalRef()s the stored representative; the Bfs - // fixture never wires that JNI slot (PollWatchedTargetsTest has its own). - jni_tbl.NewLocalRef = &mock_NewLocalRefPassthrough; - // NOTE: no liveness calls here - the Bfs tests run without them, and liveness state changes - // (setGcGenerationsForTest) alter the pass machinery's behavior; each harness test resets - // liveness explicitly where it wants it (resetLivenessForPod below). - } - - static jobject JNICALL mock_NewLocalRefPassthrough(JNIEnv *, jobject ref) { - return ref; - } - - void resetLivenessForPod() { - LivenessTracker::instance()->klassPopulationResetForTest(); - LivenessTracker::instance()->setGcGenerationsForTest(true); - LivenessTracker::instance()->leakTagPoolResetForTest(); - } - - void TearDown() override { - ReferenceChainsBfsTest::TearDown(); - // Hygiene for whatever suite runs next: clear any population this harness seeded. - LivenessTracker::instance()->klassPopulationResetForTest(); - } - - // Builds the hotdog leak topology: holder-class static -> synchronized wrapper (with its - // mutex==this self-edge) -> c -> K leak-tagged chunks, plus 2 flood classes x 16 static-held - // nodes for anchor-tier volume. - PodTopology buildLeakPod() { - PodTopology topo; - // CAPACITY FIXTURE CONTRACT: classes registered via - // classes.push_back({(void*)&node_tags[node], ...}) capture the node's ADDRESS as the - // jclass identity - std::vector growth would reallocate node_tags and dangle every captured - // pointer, silently making indexOfNode() fail for those classes in the static sweep (the - // holder array seeds no expandable class, the STATIC_FIELD edges never replay, nothing - // admits). - node_tags.reserve(600); - tags_ever_assigned.reserve(600); - topo.leak_cls_idx = addClass((void *)0x5001, "[B"); - topo.wrapper_cls_idx = addClass( - (void *)0x5002, - "Ljava/util/Collections$SynchronizedRandomAccessList;"); - topo.list_cls_idx = addClass((void *)0x5003, "Ljava/util/ArrayList;"); - - topo.holder_class_node = addNode(); - topo.wrapper_node = addNode(); - topo.list_node = addNode(); - // The holder class node doubles as the jclass identity (see - // DiscoversObjectRetainedOnlyByStaticField): register it as a loaded class so the - // static-field sweep admits the wrapper root-attached. - classes.push_back({(void *)&node_tags[topo.holder_class_node], - "Lcom/rc/pod/Holder;"}); - // Flood classes get their own jclass-identity nodes too, so their statics are separate - // anchors (not more statics on the holder). - const char *flood_sigs[2] = {"Lcom/rc/pod/FloodA;", - "Lcom/rc/pod/FloodB;"}; - for (int c = 0; c < 2; c++) { - topo.flood_class_nodes.push_back(addNode()); - classes.push_back( - {(void *)&node_tags[topo.flood_class_nodes[c]], flood_sigs[c]}); - } - - for (int i = 0; i < kChunks; i++) { - topo.chunk_nodes.push_back(addNode()); - } - for (int i = 0; i < 32; i++) { - topo.flood_nodes.push_back(addNode()); - } - // Backlog volume: a 300-edge deep chain under one flood root. - for (int i = 0; i < 300; i++) { - topo.filler_nodes.push_back(addNode()); - } - - script = { - // LEAK_BUFFER: the holder class's static field -> wrapper. - {JVMTI_HEAP_REFERENCE_STATIC_FIELD, topo.holder_class_node, - topo.wrapper_node, topo.wrapper_cls_idx}, - // The synchronized wrapper's mutex == this self-edge (round-16 fix-A shape: must not - // demote the root-attached wrapper). - {JVMTI_HEAP_REFERENCE_FIELD, topo.wrapper_node, topo.wrapper_node, - topo.wrapper_cls_idx}, - // wrapper -> c -> chunks. - {JVMTI_HEAP_REFERENCE_FIELD, topo.wrapper_node, topo.list_node, - topo.list_cls_idx}, - }; - for (int i = 0; i < kChunks; i++) { - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, topo.list_node, - topo.chunk_nodes[i], topo.leak_cls_idx}); - } - // Flood volume: each flood class holds 16 statics. - for (int c = 0; c < 2; c++) { - for (int i = 0; i < 16; i++) { - script.push_back({JVMTI_HEAP_REFERENCE_STATIC_FIELD, - topo.flood_class_nodes[c], - topo.flood_nodes[16 * c + i], - topo.leak_cls_idx /* any class; volume only */}); - } - } - // The filler chain hangs off flood node 0 (already a static anchor). - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, topo.flood_nodes[0], - topo.filler_nodes[0], topo.leak_cls_idx}); - for (int i = 0; i + 1 < (int)topo.filler_nodes.size(); i++) { - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, - topo.filler_nodes[i], topo.filler_nodes[i + 1], - topo.leak_cls_idx}); - } - reseedChunkLeakTags(topo); - return topo; - } - - // Leak tags model the poll's tagLeakInstances() assignment (the pod re-tags within minutes of - // every restart - the verify-not-retag state machine); the walk's leak-tag interception path - // consumes them for real from the live node tags. - void reseedChunkLeakTags(const PodTopology &topo) { - for (int i = 0; i < (int)topo.chunk_nodes.size(); i++) { - jlong leak_tag = 1073741824LL + 100 + i; - node_tags[topo.chunk_nodes[i]] = leak_tag; - tags_ever_assigned[topo.chunk_nodes[i]] = leak_tag; - } - } - - // Resolves the leak class's tracker-side klass id from the classTags table via the SAME tag - // value the mock passes as the chunk edges' class_tag (tags[klass_ptr] - the sweep's negative - // class tag, set during phase 1's static sweep). - u32 resolveLeakKlassId(const PodTopology &topo) { - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - EXPECT_GT(ReferenceChainsTestAccessor::passesRunForTest(), 0) - << "phase-1 pass never ran"; - jlong class_tag = tags[classes[topo.leak_cls_idx].klass]; - EXPECT_LT(class_tag, 0) << "leak class never sweep-tagged (class_tag=" - << class_tag << ")"; - return tracker->classTags()->resolve(class_tag); - } - - // Seeds the leak-side liveness (population growth + qualifying tid + representative = the first - // chunk), mirroring PollWatchedTargetsTest::seedGrowingCandidate. - void seedLeakLiveness(u32 klass_id, const PodTopology &topo) { - int slot; - bool created; - for (u16 i = 1; i <= 20; i++) { - LivenessTracker::instance()->klassPopulationRecordForTest( - klass_id, i, i, &slot, &created); - LivenessTracker::instance()->tidTrendRecordForTest( - klass_id, /*tid=*/4242, (u32)i, (u64)i); - } - LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( - nullptr, klass_id, (jweak)&node_tags[topo.chunk_nodes[0]]); - // The representative's GetTag() identity: the mock tags map is keyed by object pointer; - // getTag(rep) must report the rep's leak tag (the reseed in - // reseedChunkLeakTags() above made node_tags hold it - chunk_leak_tags - // itself is not kept). - tags[&node_tags[topo.chunk_nodes[0]]] = - node_tags[topo.chunk_nodes[0]]; - ASSERT_NE(0, tags[&node_tags[topo.chunk_nodes[0]]]) - << "representative's leak tag missing - L1/L2 assertions below " - "would silently test an untagged representative"; - } - - // One threadLoop iteration, fake clock: the exact body order from - // ReferenceChainTracker::threadLoop() (shouldRunPass -> runPass -> pollWatchedTargets, poll - // unconditional). - bool drivePodCycle(u64 &fake_now) { - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - bool should_run = tracker->shouldRunPassForTest(fake_now); - if (should_run) { - bool truncated = false; - tracker->runPass(&mock_jvmti, &mock_jni, &truncated); - } - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - fake_now += kCycleNs; - return should_run; - } - - // Seeds a throwaway candidate klass (NO instances, never matches any real class) whose only job - // is arming the leak signal so the phase-1 pass runs: the dormancy invariant (L8) proved - // shouldRunPass stays false without a candidate, and the klass-id resolution needs a pass. - void seedThrowawayLiveness(u32 klass_id) { - int slot; - bool created; - for (u16 i = 1; i <= 20; i++) { - LivenessTracker::instance()->klassPopulationRecordForTest( - klass_id, i, i, &slot, &created); - LivenessTracker::instance()->tidTrendRecordForTest( - klass_id, /*tid=*/4242, (u32)i, (u64)i); - } - LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( - nullptr, klass_id, (jweak)0xBADC0DE); - } - - // Two-phase pod bring-up. Phase 1: a throwaway candidate arms the leak signal so cycle 1's pass - // runs - the sweep tags the classes and the walk admits the wrapper subtree, which makes the - // leak class's tracker-side klass id resolvable (the only reliable source of the id is the - // system's own resolve() over a real frontier entry). - void bringUpPod(const PodTopology &topo, u32 &leak_klass_id_out, - u64 &fake_now) { - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - // Seed the fake clock from the REAL monotonic clock: the pain budgets' _last_update_ns is - // OS::nanotime()-based, so a fake epoch starting at ~0 sits before every budget's - // initialization and canStartNow() blocks every pass (the dormancy test passes either way - - // no passes at all - so only the pass-running tests caught this). - fake_now = OS::nanotime(); - resetLivenessForPod(); - seedThrowawayLiveness(/*klass_id=*/999); - // The candidate arms in the POLL, which runs after shouldRunPass in each cycle - so the - // first cycle only arms, and a pass actually runs one cycle later. - bool ran = false; - for (int i = 0; i < 6 && !ran; i++) { - ran = drivePodCycle(fake_now); - } - leak_klass_id_out = resolveLeakKlassId(topo); - LivenessTracker::instance()->klassPopulationResetForTest(); - ReferenceChainsTestAccessor::restartSearchForTest(); - reseedChunkLeakTags(topo); - seedLeakLiveness(leak_klass_id_out, topo); - // Poll once BEFORE the first pass of the new search: the candidate slot registers in the - // poll, and the pass's admission auto-mark requires the slot to exist - // (heapReferenceCallback's auto-mark guards on _candidate_count > 0). - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - } - - // Captures stdout per distinct ReferenceChainTracker/LivenessTracker line prefix (L4: the - // log-budget invariant - the round-15 800k/min flood and the round-18 tick stragglers become - // assertion failures). - struct StdoutCapture { - int saved_fd = -1; - int tmp_fd = -1; - char path[64] = {0}; - bool done = false; - StdoutCapture() { - snprintf(path, sizeof(path), "/tmp/podjar_stdout_XXXXXX"); - tmp_fd = mkstemp(path); - fflush(stdout); - saved_fd = dup(1); - dup2(tmp_fd, 1); - } - std::map perPrefixCounts() { - if (done) { - return {}; - } - done = true; - fflush(stdout); - dup2(saved_fd, 1); - close(saved_fd); - saved_fd = -1; - lseek(tmp_fd, 0, SEEK_SET); - std::map counts; - FILE *f = fdopen(tmp_fd, "r"); - char line[512]; - while (fgets(line, sizeof(line), f)) { - char cls[96], fn[96]; - if (sscanf(line, "[TEST::INFO] %95[^:]::%95[a-zA-Z]", cls, - fn) == 2) { - counts[std::string(cls) + "::" + fn]++; - } - } - fclose(f); - unlink(path); - tmp_fd = -1; - return counts; - } - }; -}; - -TEST_F(PodInAJarTest, SystemLivenessLeakChainsBuildAndCanaryResolves) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - PodTopology topo = buildLeakPod(); - u32 leak_klass_id; - u64 fake_now; - bringUpPod(topo, leak_klass_id, fake_now); - - int cycles_run = 0; - while (cycles_run < 120 && - ReferenceChainsTestAccessor::resolvedChainCountForTest() < - (size_t)kChunks && - ReferenceChainsTestAccessor::searchStateForTest() == - SearchState::RUNNING) { - drivePodCycle(fake_now); - cycles_run++; - } - - // L1: every leak-tagged chunk's tag appears as a cached chain target (leak-correlated events - - // the pod's round-16 end-goal state). - auto targets = ReferenceChainsTestAccessor::resolvedChainTargetsForTest(); - for (int i = 0; i < kChunks; i++) { - jlong leak_tag = 1073741824LL + 100 + i; - EXPECT_NE(std::find(targets.begin(), targets.end(), (u64)leak_tag), - targets.end()) - << "leak tag " << leak_tag - << " never became a cached chain target after " << cycles_run - << " cycles"; - } - - // L2: the canary resolved for the LEAK klass (found bit set on its slot - the round-19 - // criterion; fails on any pre-788d7b2a7 build). - int leak_slot = -1; - for (int s2 = 0; s2 < ReferenceChainsTestAccessor::candidateCountForTest(); - s2++) { - if (ReferenceChainsTestAccessor::candidateKlassIdForTest(s2) == - leak_klass_id) { - leak_slot = s2; - break; - } - } - ASSERT_GE(leak_slot, 0) << "leak klass never registered a candidate slot"; - EXPECT_TRUE(ReferenceChainsTestAccessor::candidateFoundBitsForTest() & - (1ULL << leak_slot)) - << "leak candidate slot never marked found despite leak-tag chains"; -} - -TEST_F(PodInAJarTest, SystemSearchCompletesNaturally) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - PodTopology topo = buildLeakPod(); - u32 leak_klass_id; - u64 fake_now; - bringUpPod(topo, leak_klass_id, fake_now); - - int cycles = 0; - while (cycles < 200 && - ReferenceChainsTestAccessor::searchStateForTest() == - SearchState::RUNNING) { - drivePodCycle(fake_now); - cycles++; - } - - // L3: a healthy-topology search must end COMPLETED (all candidates found), never ABANDONED - // (TTL/frontier-cap). - EXPECT_EQ((u8)SearchState::COMPLETED, - ReferenceChainsTestAccessor::searchStateForTest()) - << "search did not complete naturally within " << cycles - << " cycles (state=" - << (int)ReferenceChainsTestAccessor::searchStateForTest() << ")"; - EXPECT_GT(ReferenceChainsTestAccessor::passesRunForTest(), 0); -} - -TEST_F(PodInAJarTest, SystemRestartLeavesNothingBehind) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - PodTopology topo = buildLeakPod(); - u32 leak_klass_id; - u64 fake_now; - bringUpPod(topo, leak_klass_id, fake_now); - for (int i = 0; i < 120 && - ReferenceChainsTestAccessor::resolvedChainCountForTest() == 0; - i++) { - drivePodCycle(fake_now); - } - ASSERT_GT(ReferenceChainsTestAccessor::resolvedChainCountForTest(), - (size_t)0); - - ReferenceChainsTestAccessor::restartSearchForTest(); - - // L6: the restart contract as a test instead of discipline. - EXPECT_GT(ReferenceChainsTestAccessor::resolvedChainCountForTest(), - (size_t)0) - << "resolved chains should persist across restarts by design"; - EXPECT_EQ((size_t)0, - ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); - EXPECT_EQ((size_t)0, - ReferenceChainsTestAccessor::staticAnchorFreshQueueSizeForTest()); - EXPECT_TRUE(ReferenceChainsTestAccessor::anchorIndexIsEmptyForTest()); - for (int s = 0; s < 5; s++) { - EXPECT_EQ(0, - ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(s)) - << "discovered slot " << s << " survived restartSearch()"; - } -} - -TEST_F(PodInAJarTest, TopologyCapacityContractStaticAdmits) { - // The topology-builder capacity contract as a regression: the FULL buildLeakPod topology + one - // direct pass must admit the wrapper static (and with it the whole leak subtree). - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - PodTopology topo = buildLeakPod(); - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - EXPECT_GT(tags_ever_assigned[topo.wrapper_node], 0) - << "wrapper static never admitted (holder_cls_tag=" - << node_tags[topo.holder_class_node] - << " sweep_gate=" << ReferenceChainsTestAccessor::sweepGateStaticCountForTest() - << "/" << ReferenceChainsTestAccessor::sweepGateResolvedCountForTest() - << ")"; - - tracker->stop(); -} - -TEST_F(PodInAJarTest, SystemHealthyAppIsDormant) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // L8: the accidental 3h pod control, encoded - a heap with no leak candidates must produce ZERO - // searches (the candidate/generations gate holds the machinery dormant on a healthy app). - PodTopology topo = buildLeakPod(); - // No seedLeakLiveness(): no candidate ever qualifies. bringUpPod is also skipped - it seeds - // liveness; drive raw cycles instead. - (void)topo; - resetLivenessForPod(); - u64 fake_now = OS::nanotime(); - for (int i = 0; i < 20; i++) { - drivePodCycle(fake_now); - } - EXPECT_EQ(0, ReferenceChainsTestAccessor::passesRunForTest()) - << "searches ran on a healthy (candidate-less) app"; - EXPECT_EQ((size_t)0, - ReferenceChainsTestAccessor::resolvedChainCountForTest()); -} - -TEST_F(PodInAJarTest, SystemLogBudgetPerPass) { -#ifndef DEBUG - // The gtest binary compiles the main sources WITHOUT DEBUG (round-16 lesson: TEST_LOG is a - // no-op here) - the log-budget invariant can only run in a DEBUG-built test binary. - GTEST_SKIP() << "log budget needs a DEBUG-built gtest binary"; -#else - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - PodTopology topo = buildLeakPod(); - u32 leak_klass_id; - u64 fake_now; - bringUpPod(topo, leak_klass_id, fake_now); - for (int i = 0; i < 3; i++) { - drivePodCycle(fake_now); - } - - // L4: capture one full cycle (pass + poll) and bound every distinct line shape. - StdoutCapture capture; - drivePodCycle(fake_now); - auto counts = capture.perPrefixCounts(); - ASSERT_FALSE(counts.empty()) << "no diagnostics captured at level 2"; - for (const auto &kv : counts) { - EXPECT_LE(kv.second, 300) << "log line shape '" << kv.first - << "' fired " << kv.second - << " times in one pass+poll cycle"; - } -#endif -} - diff --git a/ddprof-lib/src/test/cpp/referenceChainsRotationTests.inc b/ddprof-lib/src/test/cpp/referenceChainsRotationTests.inc deleted file mode 100644 index 5790d23cc7..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsRotationTests.inc +++ /dev/null @@ -1,824 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -TEST_F(ReferenceChainsBfsTest, StaleRootAttributionUpgradesOnRediscovery) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // Synthetic stack-local root: admitted, root-attached (parent_tag == 0), its owning frame has - // since "gone away" from the design doc's scenario (nothing further to model here - the entry - // simply stays as-is until a more durable root is discovered). - jlong tag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, /*parent_tag=*/0, /*depth=*/0, - FrontierEntryState::EXPANDED, JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - - // A second, equally-or-less durable root discovery does not overwrite the recorded root_kind. - EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( - frontier, tag, JVMTI_HEAP_REFERENCE_JNI_LOCAL)); - FrontierEntry entry{}; - ASSERT_TRUE(frontier->lookup(tag, &entry)); - EXPECT_EQ(JVMTI_HEAP_REFERENCE_STACK_LOCAL, entry.root_kind); - - // A durable root (JNI global) attaching to the same object upgrades it. - EXPECT_TRUE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( - frontier, tag, JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); - ASSERT_TRUE(frontier->lookup(tag, &entry)); - EXPECT_EQ(JVMTI_HEAP_REFERENCE_JNI_GLOBAL, entry.root_kind); - EXPECT_EQ(0, entry.parent_tag); // still root-attached, unchanged - - // An even less durable root discovered afterwards cannot downgrade it. - EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( - frontier, tag, JVMTI_HEAP_REFERENCE_MONITOR)); - ASSERT_TRUE(frontier->lookup(tag, &entry)); - EXPECT_EQ(JVMTI_HEAP_REFERENCE_JNI_GLOBAL, entry.root_kind); - - tracker->stop(); -} - -// Exercises the invariant conflict that durability re-verification exists to catch: a non-root -// entry (parent_tag != 0) rediscovered as if via a root context must never have its root_kind -// overwritten - doing so would leave a non-zero root_kind on an entry nothing else treats as -// root-attached (referenceChains.h's FrontierEntry::root_kind comment), since this mutator never -// touches parent_tag. -TEST_F(ReferenceChainsBfsTest, NonRootAttachedEntryNeverUpgraded) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // Parent Y (root-attached) and child X, admitted the way frontier re-expansion admits a - // non-root child: non-root (parent_tag == Y's tag), root_kind == 0. - jlong yTag = 1; - jlong xTag = 2; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, yTag, /*parent_tag=*/0, /*depth=*/0, - FrontierEntryState::EXPANDED, JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, xTag, /*parent_tag=*/yTag, /*depth=*/1, - FrontierEntryState::EXPANDED, /*root_kind=*/0)); - - // Re-expanding Y rediscovers an edge to X (already tracked) - even if this rediscovery is - // (incorrectly) attempted with a durable root_kind, it must be rejected because X is not - // root-attached. - EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( - frontier, xTag, JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - FrontierEntry entry{}; - ASSERT_TRUE(frontier->lookup(xTag, &entry)); - EXPECT_EQ(0, entry.root_kind); - EXPECT_EQ(yTag, entry.parent_tag); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, RotationSelectsOnlyTransientExpandedRootAttachedEntries) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // Eligible: root-attached, EXPANDED, transient root_kind. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - // Not eligible: durable root_kind. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 2, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); - // Not eligible: transient but still FRONTIER, not yet EXPANDED. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 3, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_JNI_LOCAL)); - // Not eligible: transient root_kind but not root-attached. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 4, /*parent_tag=*/1, 1, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_JNI_LOCAL)); - // Eligible: root-attached, EXPANDED, transient (JNI local this time). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 5, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_JNI_LOCAL)); - - std::vector selected = - ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation(10); - std::sort(selected.begin(), selected.end()); - EXPECT_EQ((std::vector{1, 5}), selected); - - // Selected tags are queued for re-expansion, exactly like an ordinary admission would queue a - // newly-discovered tag. - EXPECT_EQ(2u, ReferenceChainsTestAccessor::priorityExpandSize()); - - tracker->stop(); -} - -// N transient-root_kind entries, rotation size R: every entry must be selected at least once within -// ceil(N/R) calls, regardless of where the cursor happened to start. -TEST_F(ReferenceChainsBfsTest, RotationCoversAllEntriesWithinCeilNOverR) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - const int N = 10; - const int R = 3; - for (jlong tag = 1; tag <= N; tag++) { - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - } - - std::unordered_set covered; - int calls = (N + R - 1) / R; - for (int i = 0; i < calls; i++) { - std::vector selected = - ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation(R); - for (jlong tag : selected) { - covered.insert(tag); - } - } - EXPECT_EQ((size_t)N, covered.size()); - - tracker->stop(); -} - -// collectStaleExpandedEntriesForRotation()'s own EXPANDED-only criterion is a strict superset of -// collectStaleRootKindEntriesForRotation()'s (which also requires parent_tag == 0 and a transient -// root_kind), and runPassManualWalk() calls the root-kind collector first, into the very same -// _priority_expand deque. -TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationDoesNotDuplicateRootKindSelection) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // Eligible for both collectors: EXPANDED, root-attached, transient root_kind - exactly the - // overlap collectStaleRootKindEntriesForRotation() will pick up first. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - // Eligible only for the EXPANDED-only sweep: EXPANDED but not root-attached. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 2, /*parent_tag=*/1, 1, FrontierEntryState::EXPANDED, - /*root_kind=*/0)); - - std::vector root_kind_selected = - ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation( - ReferenceChainsTestAccessor::rootKindRotationBudget()); - EXPECT_EQ((std::vector{1}), root_kind_selected); - - std::vector stale_expanded_selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( - ReferenceChainsTestAccessor::staleExpandedRotationBudget()); - // Tag 1 is already queued from the root-kind collector above and must not be selected again; - // tag 2 is newly discovered by this sweep. - EXPECT_EQ((std::vector{2}), stale_expanded_selected); - - std::vector queued = ReferenceChainsTestAccessor::priorityExpandContents(); - EXPECT_EQ((std::vector{1, 2}), queued); - std::unordered_set unique_queued(queued.begin(), queued.end()); - EXPECT_EQ(queued.size(), unique_queued.size()); - - tracker->stop(); -} - -// A tag left over in _priority_expand from a prior pass's truncated expandFrontier() batch (see -// expandFrontier()'s own "leave the batch at the front of the source queue for a later pass to -// retry" comment) must also be skipped by collectStaleExpandedEntriesForRotation() - not just tags -// queued by collectStaleRootKindEntriesForRotation() earlier in the same call. -TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationSkipsPreexistingQueueEntries) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - - // Simulate a truncated batch from a prior pass still sitting at the front of _priority_expand, - // without going through the root-kind collector at all - the leftover entry alone must still be - // enough to suppress a duplicate. - ReferenceChainsTestAccessor::pushPriorityExpand(1); - - std::vector stale_expanded_selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( - ReferenceChainsTestAccessor::staleExpandedRotationBudget()); - EXPECT_TRUE(stale_expanded_selected.empty()); - - std::vector queued = ReferenceChainsTestAccessor::priorityExpandContents(); - EXPECT_EQ((std::vector{1}), queued); - - tracker->stop(); -} - -// End-to-end proof of the prof-analyzer-hotdog-jb pod's actual leak shape: a static-field-rooted -// collection (like ProfileAnalyzer.LEAK_BUFFER) whose owning node is admitted and fully EXPANDED -// once, then has a *new* element appended to it afterward - mirroring a Java List field being -// mutated in place, never reassigned, well after admitStaticFieldRoots()'s one-time sweep. -TEST_F(ReferenceChainsBfsTest, RotationDiscoversLateElementOfExpandedStaticFieldCollectionWithoutSearchCompleting) { - Arguments args; - // budget=8 -> rotation_reserved_budget = min(8/2, 272) = 4, ordinary = 4: both slices non-zero, - // unlike a budget=1 pattern which would zero out rotation's reserved slice entirely (min(0, - // 272) == 0). - ASSERT_FALSE(args.parse("referencechains=true:hops=5000:budget=8:firstpassbudget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int classNode = addNode(); - int listNode = addNode(); - int seedChildNode = addNode(); - int lateChildNode = addNode(); - - // Distractor chain: a long, independently root-seeded chain that never fully drains within this - // test's bounded pass loops below, so the overall search always has forward progress available - // and never reaches SearchState::COMPLETED (nor NO_PROGRESS_PASS_LIMIT-triggered ABANDONED) - // purely as a side effect of this test's own loop bounds. - const int kDistractorNodes = 500; - std::vector distractor(kDistractorNodes); - for (int i = 0; i < kDistractorNodes; i++) { - distractor[i] = addNode(); - } - - // addClass() captures classNode's address in node_tags' backing storage - must come after every - // addNode() call above (including the distractor loop), or a later push_back reallocating - // node_tags would silently leave this pointer dangling (indexOfNode() would then never match - // it). - addClass((void *)&node_tags[classNode], "Lcom/rc/statics/GrowingListHolder;"); - - script = { - // listNode is retained only via classNode's static field - the same shape as - // DiscoversObjectRetainedOnlyByStaticField above. - {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, listNode, -1}, - // listNode's one pre-existing element, discovered the first time listNode itself is - // expanded. - {JVMTI_HEAP_REFERENCE_FIELD, listNode, seedChildNode, -1}, - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, distractor[0], -1}, - }; - for (int i = 0; i + 1 < kDistractorNodes; i++) { - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, distractor[i], distractor[i + 1], -1}); - } - - // Phase 1: run passes until listNode has been fully expanded (its one pre-existing child - // discovered), without ever letting the search complete. - bool truncated = true; - FrontierEntry listEntry{}; - bool listExpanded = false; - for (int i = 0; i < 200 && !listExpanded; i++) { - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - jlong listTag = tags_ever_assigned[listNode]; - if (listTag != 0 && tracker->frontierTable()->lookup(listTag, &listEntry) - && listEntry.state == FrontierEntryState::EXPANDED) { - listExpanded = true; - } - } - ASSERT_TRUE(listExpanded); - ASSERT_NE(0, tags_ever_assigned[seedChildNode]); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - - // Phase 2: simulate a new element appended to the leaking static field's list *after* - // listNode's one-time expansion - the exact "growing collection" shape found in the real pod's - // leak generator. - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, listNode, lateChildNode, -1}); - - for (int i = 0; i < 200 && tags_ever_assigned[lateChildNode] == 0; i++) { - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - } - - // The late element was discovered purely via rotation re-expanding listNode - and, critically, - // without the search ever completing (no dependency on a full heap walk finishing). - ASSERT_NE(0, tags_ever_assigned[lateChildNode]); - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - - std::vector chain; - ASSERT_TRUE(tracker->frontierTable()->reconstructChain( - tags_ever_assigned[lateChildNode], &chain)); - FrontierEntry lateEntry{}; - ASSERT_TRUE(tracker->frontierTable()->lookup( - tags_ever_assigned[lateChildNode], &lateEntry)); - EXPECT_EQ(tags_ever_assigned[listNode], lateEntry.parent_tag); - - tracker->stop(); -} - -// Proof of the fix for the actual prof-analyzer-hotdog-jb stall: -// collectStaleExpandedEntriesForRotation() (referenceChains.cpp) used to always rescan -// FrontierTable slots starting from tag 1, unlike its sibling -// collectStaleRootKindEntriesForRotation() which already carried its own persistent cursor. -TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationCoversHighTagEntryBehindLowTagPopulationWithinBoundedPasses) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - const int lowTagBudget = ReferenceChainsTestAccessor::staleExpandedRotationBudget(); - // Comfortably above the 256-entry cap, so the low-tag population alone would fill every sweep - // before an always-from-1 scan could ever reach the high-tag entry below - mirrors a real - // multi-GiB heap's frontier table, which accumulates far more than 256 long-lived, perpetually- - // EXPANDED entries (bootstrap classes, caches, etc.) well before any one leak-candidate class - // even loads. - const int lowTagPopulation = lowTagBudget + 50; - for (jlong tag = 1; tag <= lowTagPopulation; tag++) { - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STACK_LOCAL)); - } - - // The leak candidate's own owning node - e.g. LEAK_BUFFER's list, admitted via a static field - // only once its class loads, well after the JVM's own bootstrap population already occupies - // every low tag number. - const jlong highTag = lowTagPopulation + 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, highTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD)); - - // Several simulated passes: each iteration mirrors one real pass - - // collectStaleExpandedEntriesForRotation() runs once, then clearPriorityExpand() mirrors - // expandFrontier() having drained whatever it selected before the next pass's sweep resumes - // from the cursor. - const int table_size = lowTagPopulation + 1; - const int calls = (table_size + lowTagBudget - 1) / lowTagBudget; - bool highTagSelected = false; - std::unordered_set covered; - for (int pass = 0; pass < calls && !highTagSelected; pass++) { - std::vector selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( - lowTagBudget); - for (jlong tag : selected) { - covered.insert(tag); - if (tag == highTag) { - highTagSelected = true; - } - } - ReferenceChainsTestAccessor::clearPriorityExpand(); - } - - EXPECT_TRUE(highTagSelected) - << "highTag was never selected within ceil(table_size / max_count) " - "passes - the fix's coverage guarantee does not hold"; - EXPECT_EQ((size_t)table_size, covered.size()); - - tracker->stop(); -} - -// trackLeakAccumulation() - the admission-time hook (called from - -TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationAggregatesBySignatureAndFanout) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987; - constexpr u32 kParent1Klass = 100, kParent2Klass = 200; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - jlong parent1Tag = 1, parent2Tag = 2; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parent1Tag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParent1Klass)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parent2Tag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParent2Klass)); - - // 3 children of the watched leaf klass under parent1, 1 under parent2 - each call simulates one - // admission (the childTag argument is only used by production code for logging/future use, not - // read by this method). - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 10); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 11); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 12); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent2Tag, 20); - - EXPECT_EQ(3u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParent1Klass)); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParent2Klass)); - EXPECT_EQ(3u, ReferenceChainsTestAccessor::leakParentFanout(parent1Tag)); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parent2Tag)); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsUnwatchedKlass) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); - jlong parentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, /*class_tag=*/100)); - - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, /*class_tag=*/555, - parentTag, 10); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); - EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsRootAttachedChild) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); - - // parent_tag == 0 - a root-attached leaf itself, nothing to attribute a container to. - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, 987, /*parent_tag=*/0, 10); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsWhenParentNotFound) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); - - // parent_tag=99 was never inserted - graceful no-op, not a crash. - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, 987, /*parent_tag=*/99, 10); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); - - tracker->stop(); -} - -// Proof of the actual bug this design was found fixing: the classMap dictionary id (referrer_klass) -// for the exact same class can differ depending on which subsystem/generation resolved it (see -// class_tag's own comment, referenceChains.h, for the real-world case - "[B" resolving to two -// different classMap ids for LivenessTracker vs. -TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationMatchesByClassTagEvenWhenReferrerKlassDiffers) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafClassTag = 987; - constexpr u32 kParentClassTag = 100; - // Deliberately different, "wrong" classMap ids - simulating exactly the compaction/regeneration - // scenario that broke referrer_klass-based matching. - constexpr u32 kParentStaleReferrerKlass = 555555; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafClassTag}); - - jlong parentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, kParentStaleReferrerKlass, - kParentClassTag)); - - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafClassTag, - parentTag, 10); - - EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafClassTag, - kParentClassTag)); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); - - tracker->stop(); -} - -// collectLeakAccumulationCandidatesForRotation() - the two-tier design - -// The central discriminating test for the whole design (per the "ubiquitous common leaf class held -// by many small unrelated parents" concern this design exists to solve): a signature with a LARGE -// but FLAT total (many unrelated parents, e.g. a common leaf class scattered across a real -// classpath) must NOT outrank a signature with a SMALLER but GROWING total (the actual leak) once a -// growth history exists - retained-size-style ranking alone would pick the wrong one every time. -TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationPrioritizesGrowingSignatureOverLargeFlatOne) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987; - constexpr u32 kGrowingParentKlass = 100; // signature A: the real leak - constexpr u32 kUbiquitousParentKlass = 999; // signature B: common, but flat - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - // Signature A: one parent, growing. - jlong growingParentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, growingParentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kGrowingParentKlass)); - - // Signature B: 20 distinct, unrelated parents, each holding just 1-2 instances of the same - // common leaf klass - a much LARGER total than A, but it will not grow between passes. - constexpr int kUbiquitousParentCount = 20; - std::vector ubiquitousParentTags; - for (int i = 0; i < kUbiquitousParentCount; i++) { - jlong tag = 100 + i; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kUbiquitousParentKlass)); - ubiquitousParentTags.push_back(tag); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, tag, 1000 + i); - } - // Pass 1: A has fanout 5, B has total 20 (20 parents x 1 each) - B is larger. - for (int i = 0; i < 5; i++) { - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, - growingParentTag, 2000 + i); - } - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( - ReferenceChainsTestAccessor::leakAccumulationRotationBudget()); - - // Pass 2: B stays exactly flat (no new admissions); A grows from 5 to 8. - for (int i = 0; i < 3; i++) { - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, - growingParentTag, 3000 + i); - } - std::vector selected = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( - ReferenceChainsTestAccessor::leakAccumulationRotationBudget()); - - ASSERT_EQ(1u, selected.size()); - EXPECT_EQ(growingParentTag, selected[0]) - << "the growing signature's parent must be selected, even though " - "the flat-but-larger signature has a much bigger absolute total"; - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationRanksByFanoutWithinWinningSignature) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kParentKlass = 100; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - jlong lowFanoutTag = 1, highFanoutTag = 2; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, lowFanoutTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, highFanoutTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, lowFanoutTag, 10); - for (int i = 0; i < 5; i++) { - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, - highFanoutTag, 20 + i); - } - - std::vector selected = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); - ASSERT_EQ(2u, selected.size()); - EXPECT_EQ(highFanoutTag, selected[0]) << "higher fanout ranks first"; - EXPECT_EQ(lowFanoutTag, selected[1]); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationRespectsMaxCountAndDedup) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kParentKlass = 100; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - jlong tag1 = 1, tag2 = 2, tag3 = 3; - for (jlong tag : {tag1, tag2, tag3}) { - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, tag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, tag, 10); - } - // tag2 already queued from an earlier collector this same pass - must be skipped even though it - // qualifies structurally. - ReferenceChainsTestAccessor::pushPriorityExpand(tag2); - - std::vector selected = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( - /*max_count=*/1); - EXPECT_EQ(1u, selected.size()) << "capped at max_count"; - EXPECT_NE(tag2, selected[0]) << "already-queued tag must not be re-selected"; - - tracker->stop(); -} - -// Reversed on round-4 pod evidence (ev-leaktag-onpod-round4): the previous EXPANDED-only selection -// made the targeted tier select ZERO every pass on a live leak - the growing holders are -// un-expanded FRONTIER-state backlog entries that the starved pending lane never reaches (a 127k -// backlog at ~120-200 objects/min). -TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationSelectsUnexpandedFrontierParentAheadOfBacklog) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kParentKlass = 100; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - // Stale re-walks already sitting in the priority lane (push_back, as the other two collectors - // do). - ReferenceChainsTestAccessor::pushPriorityExpand(900); - ReferenceChainsTestAccessor::pushPriorityExpand(901); - - jlong notYetExpandedTag = 1, expandedLowFanoutTag = 2; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, notYetExpandedTag, 0, 0, FrontierEntryState::FRONTIER, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, expandedLowFanoutTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - for (int i = 0; i < 10; i++) { - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, - notYetExpandedTag, 10 + i); - } - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, - expandedLowFanoutTag, 100); - - std::vector selected = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); - ASSERT_EQ(2u, selected.size()); - EXPECT_EQ(notYetExpandedTag, selected[0]) - << "the FRONTIER-state parent qualifies and outranks the " - "lower-fanout EXPANDED one"; - EXPECT_EQ(expandedLowFanoutTag, selected[1]); - - std::vector queue = ReferenceChainsTestAccessor::priorityExpandContents(); - ASSERT_GE(queue.size(), 4u); - EXPECT_EQ(notYetExpandedTag, queue[0]) - << "the targeted un-expanded holder must JUMP the backlog, not " - "queue behind the stale re-walks"; - EXPECT_EQ(expandedLowFanoutTag, queue[1]) - << "selection order must be preserved at the head (fanout rank)"; - EXPECT_EQ(900, queue[2]); - EXPECT_EQ(901, queue[3]); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationReturnsEmptyWhenNothingHasGrownSincePreviousPass) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kParentKlass = 100; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - - jlong parentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parentTag, 10); - - // First call establishes the baseline (delta == total, since there is no prior snapshot) and - // selects it. - std::vector firstPass = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); - ASSERT_EQ(1u, firstPass.size()); - ReferenceChainsTestAccessor::clearPriorityExpand(); - - // Second call, nothing new admitted - delta is now 0 for every signature, so nothing should be - // selected. - std::vector secondPass = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); - EXPECT_TRUE(secondPass.empty()) - << "no signature grew since the previous pass's snapshot"; - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationReturnsEmptyWhenNoSignaturesTracked) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - std::vector selected = - ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); - EXPECT_TRUE(selected.empty()); - - tracker->stop(); -} - -// seedLeakAccumulationForNewlyWatchedKlass() - the cold-start fix: retroactively - -TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationPopulatesFromAlreadyAdmittedEntries) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kParentKlass = 100; - // Entries inserted directly (as if admitted by an earlier pass), with no watched klass_id set - // at all yet at insertion time - trackLeakAccumulation() was never called for any of these. - jlong parentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - for (int i = 0; i < 4; i++) { - jlong childTag = 10 + i; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, childTag, parentTag, 1, FrontierEntryState::EXPANDED, - /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); - } - ASSERT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()) - << "nothing tracked yet - trackLeakAccumulation() was never called"; - - ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); - - EXPECT_EQ(4u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParentKlass)); - EXPECT_EQ(4u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationSkipsNonMatchingAndNonExpandedEntries) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kOtherKlass = 555, kParentKlass = 100; - jlong parentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - // Wrong class - must not be counted. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 10, parentTag, 1, FrontierEntryState::EXPANDED, - /*root_kind=*/0, /*referrer_klass=*/0, kOtherKlass)); - // Right class, but still FRONTIER (not yet EXPANDED) - must not be counted. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 11, parentTag, 1, FrontierEntryState::FRONTIER, - /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); - // Right class, root-attached (no real parent) - must not be counted. - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 12, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kLeafKlass)); - - ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationComposesWithOngoingIncrementalUpdates) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - constexpr u32 kLeafKlass = 987, kParentKlass = 100; - jlong parentTag = 1; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 10, parentTag, 1, FrontierEntryState::EXPANDED, - /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); - - // Retroactive seed sees the one pre-existing child. - ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); - - // A genuinely new admission after watching starts must add on top of the retroactive baseline, - // not reset or double it. - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parentTag, 11); - - EXPECT_EQ(2u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); - EXPECT_EQ(2u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParentKlass)); - - tracker->stop(); -} - -// Smoke test simulating the hotdog pod conditions that starved BFS: - diff --git a/ddprof-lib/src/test/cpp/referenceChainsTrackerTests.inc b/ddprof-lib/src/test/cpp/referenceChainsTrackerTests.inc deleted file mode 100644 index 2bb5b2cd09..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsTrackerTests.inc +++ /dev/null @@ -1,905 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -class PollWatchedTargetsTest : public ::testing::Test { -protected: - jvmtiInterface_1_ jvmti_tbl{}; - _jvmtiEnv mock_jvmti{}; - JNINativeInterface_ jni_tbl{}; - JNIEnv_ mock_jni{}; - - std::unordered_map tags; - std::unordered_set dead_refs; // NewLocalRef returns NULL for these - - jvmtiEnv *orig_jvmti = nullptr; - static PollWatchedTargetsTest *active_fixture; - - void SetUp() override { - active_fixture = this; - ReferenceChainsTestAccessor::reset(); - LivenessTracker::instance()->klassPopulationResetForTest(); - LivenessTracker::instance()->setGcGenerationsForTest(true); - - jvmti_tbl = jvmtiInterface_1_{}; - jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; - jvmti_tbl.GetTag = &mock_GetTag; - jvmti_tbl.SetTag = &mock_SetTag; - jvmti_tbl.GetClassSignature = &mock_GetClassSignature; - jvmti_tbl.Deallocate = &mock_Deallocate; - mock_jvmti.functions = &jvmti_tbl; - orig_jvmti = VMTestAccessor::getJvmti(); - VMTestAccessor::setJvmti(&mock_jvmti); - - jni_tbl = JNINativeInterface_{}; - jni_tbl.NewLocalRef = &mock_NewLocalRef; - jni_tbl.DeleteLocalRef = &mock_DeleteLocalRef; - jni_tbl.GetObjectClass = &mock_GetObjectClass; - mock_jni.functions = &jni_tbl; - } - - void TearDown() override { - VMTestAccessor::setJvmti(orig_jvmti); - LivenessTracker::instance()->klassPopulationResetForTest(); - LivenessTracker::instance()->setGcGenerationsForTest(false); - active_fixture = nullptr; - } - - static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { - auto it = active_fixture->tags.find(object); - *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; - return JVMTI_ERROR_NONE; - } - - static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { - active_fixture->tags[object] = tag; - return JVMTI_ERROR_NONE; - } - - static jobject JNICALL mock_NewLocalRef(JNIEnv *, jobject ref) { - if (active_fixture->dead_refs.count(ref) > 0) { - return nullptr; - } - return ref; // identity passthrough - see this fixture's own comment - } - - static void JNICALL mock_DeleteLocalRef(JNIEnv *, jobject) { - // no-op: this fixture's fake jobject values are not real JNI refs. - } - - // pollWatchedTargets()'s diagnostic class-name lookup on the candidate's representative: this - // fixture's fake jobjects carry no real class identity, so a fixed non-null jclass plus a fixed - // signature is all GetObjectClass()/GetClassSignature() need to return for that lookup to - // complete without touching a real JVM. - static jclass JNICALL mock_GetObjectClass(JNIEnv *, jobject) { - return (jclass)0xC1A55; - } - - static jvmtiError JNICALL mock_GetClassSignature(jvmtiEnv *, jclass, - char **signature_ptr, - char **generic_ptr) { - *signature_ptr = strdup("Ltest/FakeKlass;"); - if (generic_ptr != nullptr) { - *generic_ptr = nullptr; - } - return JVMTI_ERROR_NONE; - } - - static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { - free(mem); - return JVMTI_ERROR_NONE; - } - - // Seeds LivenessTracker's real population table with a growing series for `klass_id` (20 - // strictly-increasing samples - satisfies selectLeakCandidates()'s min-fill, growth/floor - // magnitude, and sustained-trend hysteresis requirements, livenessTracker.h; 20 rather than the - // 10-sample minimum fill leaves comfortable margin past the hysteresis threshold rather than - // sitting exactly on its boundary) and points its representative at `rep`. - void seedGrowingCandidate(u32 klass_id, jweak rep) { - int slot; - bool created; - for (u16 i = 1; i <= 20; i++) { - LivenessTracker::instance()->klassPopulationRecordForTest( - klass_id, i, i, &slot, &created); - // Per-(klass, tid) qualification: selectLeakCandidates() also requires a qualifying - // allocating thread. - LivenessTracker::instance()->tidTrendRecordForTest( - klass_id, /*tid=*/4242, (u32)i, (u64)i); - } - LivenessTracker::instance()->klassPopulationSetRepresentativeForTest(nullptr, klass_id, rep); - } -}; - -PollWatchedTargetsTest *PollWatchedTargetsTest::active_fixture = nullptr; - -TEST_F(PollWatchedTargetsTest, EmitsEventForAlreadyDiscoveredCandidate) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); - - // Model "already discovered by an ordinary runPass()": a root-level FrontierTable entry plus a - // matching GetTag() result, mirroring referenceChainJfrRoundtrip_ut.cpp's seeding style. - ASSERT_TRUE(tracker->frontierTable()->insert( - /*tag=*/7, /*parent_tag=*/0, /*referrer_klass=*/1, /*depth=*/0, - FrontierEntryState::EDGE)); - tags[obj] = 7; - // With class-tag matching, _candidate_frontier_tags must be set so buildCanaryChainEvent() can - // reconstruct the chain. - ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); - EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(7)); - - tracker->stop(); -} - -TEST_F(PollWatchedTargetsTest, NoEventForNotYetDiscoveredCandidate) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); - // GetTag() reports 0 (default) - no pass has reached this object yet. - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); - - tracker->stop(); -} - -TEST_F(PollWatchedTargetsTest, NoDuplicateOnRepeatPoll) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); - - ASSERT_TRUE(tracker->frontierTable()->insert( - 7, 0, 1, 0, FrontierEntryState::EDGE)); - tags[obj] = 7; - ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); - - // Klass 1 is still flagged (LivenessTracker's ranking doesn't know an event was already emitted - // for it) - a second, third, ... - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); - EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(7)); - - tracker->stop(); -} - -TEST_F(PollWatchedTargetsTest, SkipsCandidateWhoseWeakReferenceDied) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); - dead_refs.insert(obj); // NewLocalRef(rep) -> NULL, as if GC'd - - ASSERT_TRUE(tracker->frontierTable()->insert( - 7, 0, 1, 0, FrontierEntryState::EDGE)); - tags[obj] = 7; // would resolve to a discovered tag, if it could resolve - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); - - tracker->stop(); -} - -// A cached chain must not re-emit forever: once the klass's representative stops resolving -// (collected, or LRU-evicted from LivenessTracker's population table - -// klassPopulationSetRepresentativeForTest()'s ref is the stand-in for either), the very next poll -// must prune it from _resolved_chains rather than leaving a dump keep re-emitting a chain for a -// sample that is gone (see _resolved_chains' own comment, referenceChains.h, and -// pollWatchedTargets()'s "candidate died, or was evicted" branch). -TEST_F(PollWatchedTargetsTest, ChainPersistsAfterRepresentativeDies) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); - - ASSERT_TRUE(tracker->frontierTable()->insert( - 7, 0, 1, 0, FrontierEntryState::EDGE)); - tags[obj] = 7; - ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); - ASSERT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); - - // The representative died. Per-instance caching: the chain persists (it describes a reference - // path that was valid at resolution time). - dead_refs.insert(obj); - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) - << "per-instance chains persist after representative dies; " - "they expire on search restart, not on representative death"; - EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); - - tracker->stop(); -} - -// Bounded canary parent-chain walk: a cyclic/corrupt parent chain must fail -// safe instead of spinning the poll thread (the bound mirrors -// FrontierTable::reconstructChain()'s maxCapacity() bound), and legitimate -// deep chains must still reconstruct fully - the bound must not cost -// convergence. -TEST_F(PollWatchedTargetsTest, CanaryChainWalkFailsSafeOnCyclicParentChain) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - FrontierTable *frontier = tracker->frontierTable(); - // Corrupt chain: 1 -> 2 -> 1 (a cycle no legitimate BFS admission could - // produce, but insert() does not validate parent chains - exactly the - // failure mode reconstructChain()'s own bound guards against). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 2, 2, FrontierEntryState::EDGE, /*root_kind=*/0, - /*referrer_klass=*/11)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 2, 1, 3, FrontierEntryState::EDGE, /*root_kind=*/0, - /*referrer_klass=*/22)); - - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 99); - ReferenceChainsTestAccessor::setCandidateParentTagForTest(0, 1); - ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 2); - ReferenceChainsTestAccessor::setCandidateReferrerKlassForTest(0, 99); - ReferenceChainsTestAccessor::setCandidateDepthForTest(0, 4); - - ReferenceChainEvent event; - // The unbounded predecessor of this walk spun forever here; the bound - // must make it return promptly (the test itself is the termination - // proof - a regression to an unbounded walk hangs this test). - EXPECT_FALSE(ReferenceChainsTestAccessor::buildCanaryChainEventForTest( - 0, &event)); - - tracker->stop(); -} - -TEST_F(PollWatchedTargetsTest, CanaryChainWalkStillReconstructsDeepChains) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - FrontierTable *frontier = tracker->frontierTable(); - // A 200-hop linear chain, root-attached at tag 1 (parent_tag == 0): far - // past the 64-hop default _hop_cap, but the walk bounds at the FRONTIER's - // maxCapacity() (65536 by default), not the BFS hop cap - legitimate deep - // chains reconstruct exactly as before the bound. - const int kDepth = 200; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EDGE, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/1)); - for (int i = 2; i <= kDepth; i++) { - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, i, i - 1, i - 1, FrontierEntryState::EDGE, /*root_kind=*/0, - /*referrer_klass=*/(u32)i)); - } - - ReferenceChainsTestAccessor::setCandidateCountForTest(1); - ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 999); - ReferenceChainsTestAccessor::setCandidateParentTagForTest(0, kDepth); - ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 1); - ReferenceChainsTestAccessor::setCandidateReferrerKlassForTest(0, 999); - ReferenceChainsTestAccessor::setCandidateDepthForTest(0, kDepth); - - ReferenceChainEvent event; - ASSERT_TRUE(ReferenceChainsTestAccessor::buildCanaryChainEventForTest( - 0, &event)); - // Candidate's own klass first, then the reversed walk: root-side hop - // (tag 1, klass 1) first, parent-side hop (tag 200, klass 200) last. - ASSERT_EQ((size_t)(kDepth + 1), event._hops.size()); - EXPECT_EQ(999u, event._hops[0].klass_id); - for (int i = 1; i <= kDepth; i++) { - EXPECT_EQ((u32)i, event._hops[i].klass_id) - << "hop " << i; - } - EXPECT_EQ(1u, event._target_tag); // the candidate's frontier tag - EXPECT_EQ((u32)kDepth, event._depth); - // The root kind describes the chain's ROOT (the root-attached tag-1 entry - // this walk terminates at), not the candidate-side parent entry - same - // terminal-root semantics reconstructChain() uses. - EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, event._root_kind); - - tracker->stop(); -} - -TEST_F(PollWatchedTargetsTest, NoOpWhenGcGenerationsDisabled) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // Overrides this fixture's own SetUp() default - exercises the pollWatchedTargets() guard - // covering LivenessTracker's own _gc_generations gate (population tracking's own gate), not - // just this tracker's own _enabled. - LivenessTracker::instance()->setGcGenerationsForTest(false); - - int fake_object_storage = 0; - jobject obj = reinterpret_cast(&fake_object_storage); - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); - ASSERT_TRUE(tracker->frontierTable()->insert( - 7, 0, 1, 0, FrontierEntryState::EDGE)); - tags[obj] = 7; - - tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); - - EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); - - tracker->stop(); -} - -// Resolved-chain cache (ReferenceChainTracker::cacheResolvedChain()/ - -class ResolvedChainCacheTest : public ::testing::Test { -protected: - void SetUp() override { - ReferenceChainsTestAccessor::reset(); - } - - void TearDown() override { - ReferenceChainsTestAccessor::reset(); - } - - static ReferenceChainEvent makeEvent(u64 target_tag) { - ReferenceChainEvent event; - event._target_tag = target_tag; - event._depth = 0; - return event; - } -}; - -// The defining property of the "stick around" model: a cached chain is re-emitted on every dump, -// not drained once. -TEST_F(ResolvedChainCacheTest, SnapshotReEmitsOnEveryDumpWithoutClearing) { - ReferenceChainsTestAccessor::cacheChain(/*source_tag=*/1, makeEvent(7), - /*source_tag=*/7, /*search_ns=*/0); - - std::vector firstDump; - ReferenceChainsTestAccessor::drain(&firstDump); - ASSERT_EQ(1u, firstDump.size()); - EXPECT_EQ(7u, firstDump[0]._target_tag); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) - << "drain must not clear the cache"; - - // A second dump with nothing changed re-emits the same chain. - std::vector secondDump; - ReferenceChainsTestAccessor::drain(&secondDump); - ASSERT_EQ(1u, secondDump.size()); - EXPECT_EQ(7u, secondDump[0]._target_tag); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); -} - -// Re-resolving the same klass (a restart re-tags its sample, or a fresh walk finds a deeper path) -// refreshes its single cache slot in place rather than accumulating duplicates - so a dump re-emits -// one current chain per klass, not one per resolution. -TEST_F(ResolvedChainCacheTest, RefreshReplacesSameKlassInPlace) { - ReferenceChainsTestAccessor::cacheChain(1, makeEvent(7), 7, 0); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); - EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(1)); - - // Same klass, rebuilt from a new tag (e.g. after a search restart). - ReferenceChainsTestAccessor::cacheChain(1, makeEvent(9), 9, 0); - EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) - << "refresh must overwrite, not append"; - EXPECT_EQ(9, ReferenceChainsTestAccessor::resolvedChainSourceTag(1)); - - std::vector dump; - ReferenceChainsTestAccessor::drain(&dump); - ASSERT_EQ(1u, dump.size()); - EXPECT_EQ(9u, dump[0]._target_tag); -} - -// Distinct klasses each get their own slot and all re-emit together in one dump (order is -// unspecified - the cache is a map keyed by klass_id). -TEST_F(ResolvedChainCacheTest, MultipleKlassesAllSnapshotTogether) { - ReferenceChainsTestAccessor::cacheChain(1, makeEvent(1), 1, 0); - ReferenceChainsTestAccessor::cacheChain(2, makeEvent(2), 2, 0); - ReferenceChainsTestAccessor::cacheChain(3, makeEvent(3), 3, 0); - ASSERT_EQ(3u, ReferenceChainsTestAccessor::resolvedChainCount()); - - std::vector dump; - ReferenceChainsTestAccessor::drain(&dump); - ASSERT_EQ(3u, dump.size()); - std::set tags; - for (const auto &e : dump) { - tags.insert(e._target_tag); - } - EXPECT_EQ((std::set{1, 2, 3}), tags); -} - -// A brand-new klass arriving with the cache already at MAX_RESOLVED_CHAINS is dropped (and counted -// via REFERENCE_CHAIN_EVENTS_DROPPED, this codebase's own "dropped-event-without-counter" review -// lens) rather than evicting some other still-live sample's chain - but refreshing a klass that is -// already cached still succeeds even at capacity. -TEST_F(ResolvedChainCacheTest, OverflowDropsNewKlassButAllowsRefresh) { - const int cap = ReferenceChainsTestAccessor::maxResolvedChains(); - long long droppedBefore = Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED); - - for (int i = 0; i < cap; i++) { - ReferenceChainsTestAccessor::cacheChain((jlong)i, makeEvent((jlong)i), - (jlong)i, 0); - } - ASSERT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()); - EXPECT_EQ(droppedBefore, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)) - << "filling exactly to capacity must not drop anything yet"; - - // A brand-new klass at capacity is dropped and counted. - ReferenceChainsTestAccessor::cacheChain((jlong)cap, makeEvent((jlong)cap), - (jlong)cap, 0); - EXPECT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()) - << "cache must stay capped, not grow past MAX_RESOLVED_CHAINS"; - EXPECT_EQ(droppedBefore + 1, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)); - EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag((u32)cap)); - - // Refreshing an already-cached klass at capacity must still succeed - it reuses that klass's - // existing slot rather than needing a free one. - ReferenceChainsTestAccessor::cacheChain(/*source_tag=*/0, makeEvent(999), - /*source_tag=*/999, 0); - EXPECT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()); - EXPECT_EQ(999, ReferenceChainsTestAccessor::resolvedChainSourceTag(0)); - EXPECT_EQ(droppedBefore + 1, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)) - << "an in-place refresh must not count as a drop"; -} - -// Pause-time pacing controller: pause-time-SLO feedback loop - -TEST_F(ReferenceChainsTest, PacingHoldsSteadyWhenPassesLandExactlyOnCeiling) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int startBudget = ReferenceChainsTestAccessor::effectiveBudget(); - u64 startCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); - ASSERT_EQ(4000, startBudget); // starts pinned at the configured ceiling - - // A pass landing exactly on the pause-time target is a zero error every call - the controller - // should never move away from its starting point, regardless of how many such passes are - // observed in a row. - for (int i = 0; i < 10; i++) { - ReferenceChainsTestAccessor::updatePacing(5 * 1000000ULL); // 5ms - EXPECT_EQ(startBudget, ReferenceChainsTestAccessor::effectiveBudget()); - EXPECT_EQ(startCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); - } - - tracker->stop(); -} - -TEST_F(ReferenceChainsTest, PacingShrinksBudgetAndWidensCadenceWhenOverCeiling) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int initialBudget = ReferenceChainsTestAccessor::effectiveBudget(); - u64 initialCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); - - // A pass taking 10x the pause-time ceiling, fed repeatedly (a constant input - the plan's own - // "does not oscillate indefinitely" scenario). - int lastBudget = initialBudget; - u64 lastCadence = initialCadence; - for (int i = 0; i < 20; i++) { - ReferenceChainsTestAccessor::updatePacing(50 * 1000000ULL); // 50ms - int budget = ReferenceChainsTestAccessor::effectiveBudget(); - u64 cadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); - EXPECT_LE(budget, lastBudget); // never grows while still over ceiling - EXPECT_GE(cadence, lastCadence); // never shrinks while still over ceiling - lastBudget = budget; - lastCadence = cadence; - } - - // Moved in the correct direction... - EXPECT_LT(lastBudget, initialBudget); - EXPECT_GT(lastCadence, initialCadence); - // ...and converged to a fixed point rather than oscillating: one more identical input produces - // no further change. - ReferenceChainsTestAccessor::updatePacing(50 * 1000000ULL); - EXPECT_EQ(lastBudget, ReferenceChainsTestAccessor::effectiveBudget()); - EXPECT_EQ(lastCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsTest, PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // Start from a controlled below-ceiling/above-baseline point (as if an earlier over-ceiling run - // had already shrunk/widened them - see the previous test) with a freshly reset controller, - // rather than chaining directly off a constant-input sequence like the previous test's own: - // _pause_pid's integral state would otherwise still be recovering from that sequence's windup - // for many iterations after switching to a smaller-magnitude error, muddying this test's - // per-step "moves in the correct direction every step" assertions with a transient this test is - // not about. - ReferenceChainsTestAccessor::setEffectiveBudget(2400); - ReferenceChainsTestAccessor::setEffectiveCadenceNs( - 2 * ReferenceChainsTestAccessor::baselineCadenceNs()); - ReferenceChainsTestAccessor::resetPacingController(); - int shrunkBudget = ReferenceChainsTestAccessor::effectiveBudget(); - u64 widenedCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); - - // Now feed passes comfortably under the ceiling, repeatedly (a constant input, to check - // convergence rather than oscillation). - int lastBudget = shrunkBudget; - u64 lastCadence = widenedCadence; - for (int i = 0; i < 200; i++) { - ReferenceChainsTestAccessor::updatePacing(0); // effectively instant - int budget = ReferenceChainsTestAccessor::effectiveBudget(); - u64 cadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); - EXPECT_GE(budget, lastBudget); // never shrinks while comfortably under - EXPECT_LE(cadence, lastCadence); // never widens while comfortably under - lastBudget = budget; - lastCadence = cadence; - } - - // Moved in the correct direction... and, since 50 identical comfortably-under-target passes is - // well past BORROW_WARMUP_PASSES, past the configured ceiling too - budget-borrowing lets it - // converge at the borrowed ceiling (configured budget * multiplier) instead of stalling at the - // plain configured budget. - EXPECT_GT(lastBudget, shrunkBudget); - EXPECT_EQ(4000 * ReferenceChainsTestAccessor::borrowCeilingMultiplier(), lastBudget); - EXPECT_LT(lastCadence, widenedCadence); - // ...and converged: one more identical input produces no further change. - ReferenceChainsTestAccessor::updatePacing(0); - EXPECT_EQ(lastBudget, ReferenceChainsTestAccessor::effectiveBudget()); - EXPECT_EQ(lastCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsTest, MaybeRevokeBorrowForRootEnumPassPreservesBorrowAtBoundary) { - Arguments args; - // BORROW_UNDER_TARGET_FRACTION (referenceChains.h) is 0.5, so with pausetarget=10 the - // comfortably-under-target boundary is exactly 5ms. - ASSERT_FALSE(args.parse("referencechains=true:budget=1000:pausetarget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - ReferenceChainsTestAccessor::setBorrowedBudget(500); - ReferenceChainsTestAccessor::setConsecutiveUnderTargetPasses(5); - - // Exactly at the boundary: comfortably_under_target's `<=` check must still treat this as - // comfortably under, so the borrow is preserved. - ReferenceChainsTestAccessor::maybeRevokeBorrowForRootEnumPass(5 * 1000000ULL); - EXPECT_EQ(500, ReferenceChainsTestAccessor::borrowedBudget()); - EXPECT_EQ(5, ReferenceChainsTestAccessor::consecutiveUnderTargetPasses()); - - tracker->stop(); -} - -TEST_F(ReferenceChainsTest, MaybeRevokeBorrowForRootEnumPassRevokesJustPastBoundary) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:budget=1000:pausetarget=10")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - ReferenceChainsTestAccessor::setBorrowedBudget(500); - ReferenceChainsTestAccessor::setConsecutiveUnderTargetPasses(5); - ReferenceChainsTestAccessor::setEffectiveBudget(1500); // as if borrow had raised the ceiling - - // Just past the boundary: no longer comfortably under target, so the grant is revoked - // immediately, including re-clamping _effective_budget down to the plain (non-borrowed) budget - // rather than leaving it borrow-inflated until the next ordinary pass's updatePacing() call. - ReferenceChainsTestAccessor::maybeRevokeBorrowForRootEnumPass(6 * 1000000ULL); - EXPECT_EQ(0, ReferenceChainsTestAccessor::borrowedBudget()); - EXPECT_EQ(0, ReferenceChainsTestAccessor::consecutiveUnderTargetPasses()); - EXPECT_EQ(1000, ReferenceChainsTestAccessor::effectiveBudget()); - - tracker->stop(); -} - -// PainBudget (painBudget.h) - standalone, no ReferenceChainTracker singleton - -TEST(PainBudgetTest, ClearBeforeAnythingIsEverSpent) { - PainBudget budget(0.01); - EXPECT_TRUE(budget.canStartNow(1000)); -} - -TEST(PainBudgetTest, SpendCreatesDebtThatBlocksAnImmediateSecondCall) { - PainBudget budget(0.01); // 1% - ASSERT_TRUE(budget.canStartNow(1000)); // establishes the drain baseline - budget.spend(100); // 100ms of debt - // No time has elapsed since the baseline call above - the debt cannot have drained at all yet. - EXPECT_FALSE(budget.canStartNow(1000)); -} - -TEST(PainBudgetTest, DebtDrainsProportionallyToElapsedTimeAndRefillRate) { - PainBudget budget(0.01); // 1% -> 1ms of debt needs 100ms elapsed to clear - ASSERT_TRUE(budget.canStartNow(0)); - budget.spend(10); // 10ms of debt -> needs 1000ms elapsed to fully clear - EXPECT_FALSE(budget.canStartNow(500ULL * 1000000ULL)); // 500ms elapsed - not enough - EXPECT_TRUE(budget.canStartNow(1500ULL * 1000000ULL)); // 1500ms total - enough -} - -TEST(PainBudgetTest, ZeroRefillRateNeverClearsDebt) { - PainBudget budget(0.0); - ASSERT_TRUE(budget.canStartNow(0)); - budget.spend(1); - // An enormous elapsed time still drains nothing at a 0 refill rate. - EXPECT_FALSE(budget.canStartNow(1000000000000ULL)); -} - -// Search restart (referenceChains.h's own header comment: gating a - -class SearchRestartTest : public ::testing::Test { -protected: - jvmtiInterface_1_ jvmti_tbl{}; - _jvmtiEnv mock_jvmti{}; - jvmtiEnv *orig_jvmti = nullptr; - - void SetUp() override { - ReferenceChainsTestAccessor::reset(); - LivenessTracker::instance()->klassPopulationResetForTest(); - LivenessTracker::instance()->setGcGenerationsForTest(false); - - jvmti_tbl = jvmtiInterface_1_{}; - jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; - jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; - jvmti_tbl.FollowReferences = &mock_FollowReferences; - jvmti_tbl.IterateOverReachableObjects = &mock_IterateOverReachableObjects; - jvmti_tbl.GetAvailableProcessors = &mock_GetAvailableProcessors; - mock_jvmti.functions = &jvmti_tbl; - orig_jvmti = VMTestAccessor::getJvmti(); - VMTestAccessor::setJvmti(&mock_jvmti); - } - - void TearDown() override { - VMTestAccessor::setJvmti(orig_jvmti); - LivenessTracker::instance()->klassPopulationResetForTest(); - LivenessTracker::instance()->setGcGenerationsForTest(false); - // UrgentOOMProjectionBypassesCandidateGate below sets this - reset it here (TearDown always - // runs, even after a fatal ASSERT_* return) rather than as a trailing statement in that - // test body, so a failed assertion can't leak a stale max-heap value into the next test - // sharing this singleton. - LivenessTracker::instance()->setMaxHeapBytesForTest(-1); - } - - // No loaded classes to resolve - resolveLoadedClasses() reports 0 and does nothing further. - static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count, - jclass **out) { - *count = 0; - *out = nullptr; - return JVMTI_ERROR_NONE; - } - - // ReferenceChainTracker::start() -> autoTuneDefaults() queries this whenever LivenessTracker - // reports a max heap > 0 - which UrgentOOMProjectionBypassesCandidateGate below sets. - static jvmtiError JNICALL mock_GetAvailableProcessors(jvmtiEnv *, - jint *nprocs) { - *nprocs = 1; - return JVMTI_ERROR_NONE; - } - - // Never invokes the callback - models a heap with nothing reachable from any root, so the very - // first pass completes immediately (0 admitted edges, not truncated). - static jvmtiError JNICALL mock_FollowReferences( - jvmtiEnv *, jint, jclass, jobject, const jvmtiHeapCallbacks *, - const void *) { - return JVMTI_ERROR_NONE; - } - - // runPassManualWalk()'s root enumeration - never invokes the root callback, same "nothing - // reachable from any root" heap model as mock_FollowReferences() above, so the first pass still - // completes immediately with 0 admitted edges. - static jvmtiError JNICALL mock_IterateOverReachableObjects( - jvmtiEnv *, jvmtiHeapRootCallback, jvmtiStackReferenceCallback, - jvmtiObjectReferenceCallback, const void *) { - return JVMTI_ERROR_NONE; - } - - // Same seeding helper as PollWatchedTargetsTest above (20 strictly- increasing samples - - // satisfies selectLeakCandidates()'s min-fill, growth/floor magnitude, and sustained-trend - // hysteresis requirements). - void seedGrowingCandidate(u32 klass_id, jweak rep) { - int slot; - bool created; - for (u16 i = 1; i <= 20; i++) { - LivenessTracker::instance()->klassPopulationRecordForTest( - klass_id, i, i, &slot, &created); - // Per-(klass, tid) qualification: selectLeakCandidates() also requires a qualifying - // allocating thread. - LivenessTracker::instance()->tidTrendRecordForTest( - klass_id, /*tid=*/4242, (u32)i, (u64)i); - } - LivenessTracker::instance()->klassPopulationSetRepresentativeForTest(nullptr, klass_id, rep); - } -}; - -TEST_F(SearchRestartTest, WithoutGenerationsSignalRestartStaysUnconditional) { - // gc_generations off (this fixture's SetUp default): canAffordNewSearch() has no candidate - // signal to gate on at all, so a terminal search is immediately eligible to restart - preserves - // this tracker's pre-restart behavior for a referencechains-without-generations setup (this - // class's own header comment, last paragraph). - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - - tracker->stop(); -} - -TEST_F(SearchRestartTest, GenerationsEnabledButNoCandidateBlocksFirstSearch) { - // A brand-new tracker must not pay for the initial whole-heap walk/tagging pass either when - // there is no leak candidate yet - shouldRunPass()'s !_search_started branch now shares - // canAffordNewSearch() with the restart gate below (this class's own header comment). - LivenessTracker::instance()->setGcGenerationsForTest(true); - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(1)); - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - EXPECT_EQ(0, tracker->passesRun()); - - int fake_object_storage = 0; - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); - - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(2)); - - tracker->stop(); -} - -TEST_F(SearchRestartTest, GenerationsEnabledButNoCandidateBlocksRestart) { - LivenessTracker::instance()->setGcGenerationsForTest(true); - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - - // No leak candidate flagged - nothing to justify the cost of a restart. - EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(1)); - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - - tracker->stop(); -} - -TEST_F(SearchRestartTest, RestartsOnceACandidateAppearsAndResetsPerSearchState) { - LivenessTracker::instance()->setGcGenerationsForTest(true); - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - ASSERT_EQ(1, tracker->passesRun()); - - int fake_object_storage = 0; - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); - - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); // restartSearch() runs inline - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - EXPECT_EQ(0, tracker->passesRun()); // restartSearch() zeroed per-search state - - // The next runPass() call takes the "first pass of a search" branch again, exactly like a - // brand-new tracker. - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - EXPECT_EQ(1, tracker->passesRun()); - - tracker->stop(); -} - -TEST_F(SearchRestartTest, PainBudgetBlocksARestartUntilItDrains) { - LivenessTracker::instance()->setGcGenerationsForTest(true); - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:painbudget=1")); // 1% - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int fake_object_storage = 0; - seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); - - // First-ever search: called via runPass() directly here, bypassing shouldRunPass()'s - // canAffordNewSearch() gate entirely - the candidate seeded above would satisfy that gate - // anyway (this class's own header comment). - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - - // Restart #1: _safepoint_pain_budget has never had anything spent into it yet, so this is - // always immediately affordable regardless of this first search's own cost - the cost a search - // incurs only debits the *next* restart's affordability (restartSearch()'s own spend-then-reset - // order), not its own. - ASSERT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - - // Pretend this second search cost 1000ms of safepoint time - a mocked FollowReferences call in - // this fixture takes ~0 real wall-clock time, so this accessor stands in for what a real, - // expensive pass would have accumulated into _search_pain_ms on its own. - ReferenceChainsTestAccessor::setSearchPainMs(1000); - - // Restart #2: the terminal gate charges the finished search's OWN 1000ms cost BEFORE checking - // affordability (canAffordNewSearch() must see the cost of the search that just ended, or an - // expensive search would earn one free immediate successor). - EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(2)); - EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); - - // Well past the drain point - the debt has cleared, restart #2 proceeds. - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1ULL + 200000000000ULL)); - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); - ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); - - tracker->stop(); -} - -// hasLeakSignal()'s OOM_URGENT_THRESHOLD_S fast path (referenceChains.h/.cpp): a heap-wide leak -// growing fast enough to project exhaustion sooner than the threshold must start a search -// immediately, without waiting for any klass to clear selectLeakCandidates()'s own per-klass -// ring-fill/hysteresis gate - this is the aggressive-leak gap -// GenerationsEnabledButNoCandidateBlocksFirstSearch above documents for the non-urgent case. -TEST_F(SearchRestartTest, UrgentOOMProjectionBypassesCandidateGate) { - LivenessTracker::instance()->setGcGenerationsForTest(true); - constexpr u64 SEC_NS = 1000000000ULL; - constexpr u64 MiB = 1ULL << 20; - // Same worked example as livenessTracker_ut.cpp's - // SecondsToOOMTest.RisingFloorProjectsExpectedSeconds: 700MiB rise over 7s against a 2800MiB - // max heap projects to 10s - comfortably under OOM_URGENT_THRESHOLD_S (5 minutes). - LivenessTracker::instance()->setMaxHeapBytesForTest((jlong)(2800 * MiB)); - for (int i = 0; i < 10; i++) { - LivenessTracker::instance()->heapFloorRecordForTest( - 1000 * MiB + (u64)i * 100 * MiB, (u64)i * SEC_NS); - } - - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); - EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); - - tracker->stop(); -} - -// Durability re-verification (correctness hardening). - diff --git a/ddprof-lib/src/test/cpp/referenceChainsTraversalTests.inc b/ddprof-lib/src/test/cpp/referenceChainsTraversalTests.inc deleted file mode 100644 index 361944bbac..0000000000 --- a/ddprof-lib/src/test/cpp/referenceChainsTraversalTests.inc +++ /dev/null @@ -1,606 +0,0 @@ -/* - * Copyright 2026, Datadog, Inc. - * SPDX-License-Identifier: Apache-2.0 - */ - -TEST_F(ReferenceChainsBfsTest, RollingResumePopsProcessedEntriesOnTruncatedBatch) { - Arguments args; - // budget=4: small enough that expand truncates mid-batch after admitting a few children. - ASSERT_FALSE(args.parse( - "referencechains=true:hops=5000:budget=4:firstpassbudget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - FrontierTable *frontier = tracker->frontierTable(); - - // A static-field root: classNode -> listNode (the leaking collection). - int classNode = addNode(); - int listNode = addNode(); - - // A chain of 20 children hanging off listNode. With budget=4, the callback admits 4 children - // then returns JVMTI_VISIT_ABORT (BUDGET_EXHAUSTED), truncating mid-batch. - constexpr int kChainLen = 20; - std::vector chainNodes(kChainLen); - for (int i = 0; i < kChainLen; i++) { - chainNodes[i] = addNode(); - } - - // Distractor roots: 20 independent JNI-global roots, each with one child. - constexpr int kDistractors = 20; - std::vector distractorRoots(kDistractors); - std::vector distractorChildren(kDistractors); - for (int i = 0; i < kDistractors; i++) { - distractorRoots[i] = addNode(); - distractorChildren[i] = addNode(); - } - - // addClass() must come after all addNode() calls. - addClass((void *)&node_tags[classNode], "Lcom/rc/SmokeTestHolder;"); - - script = { - {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, listNode, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, listNode, chainNodes[0], -1}, - }; - for (int i = 0; i + 1 < kChainLen; i++) { - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, chainNodes[i], chainNodes[i + 1], -1}); - } - for (int i = 0; i < kDistractors; i++) { - script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, distractorRoots[i], -1}); - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, distractorRoots[i], distractorChildren[i], -1}); - } - - // Phase 1: run passes until listNode is admitted via the static-field sweep. - bool truncated = true; - jlong listTag = 0; - for (int i = 0; i < 200 && listTag == 0; i++) { - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - listTag = tags_ever_assigned[listNode]; - } - ASSERT_NE(0, listTag) << "listNode was never admitted to the frontier"; - - // Phase 2: run passes until listNode is expanded (rolling resume pops it). - for (int i = 0; i < 200; i++) { - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - FrontierEntry entry{}; - if (frontier->lookup(listTag, &entry) && - entry.state == FrontierEntryState::EXPANDED) { - break; - } - } - FrontierEntry listEntry{}; - ASSERT_TRUE(frontier->lookup(listTag, &listEntry)); - EXPECT_EQ(FrontierEntryState::EXPANDED, listEntry.state) - << "listNode should be EXPANDED after rolling resume popped it"; - - // Verify some chain children were admitted. - int admittedChildren = 0; - for (int i = 0; i < kChainLen; i++) { - if (tags_ever_assigned[chainNodes[i]] != 0) admittedChildren++; - } - EXPECT_GT(admittedChildren, 0) - << "No chain children were admitted — expand never ran"; - - // Phase 3: run more passes until all chain children are admitted. - for (int i = 0; i < 500 && admittedChildren < kChainLen; i++) { - ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - admittedChildren = 0; - for (int j = 0; j < kChainLen; j++) { - if (tags_ever_assigned[chainNodes[j]] != 0) admittedChildren++; - } - } - EXPECT_EQ(kChainLen, admittedChildren) - << "Not all chain children were admitted within bounded passes"; - - tracker->stop(); -} - -// Verify the AIMD adaptive batch_size: with the per-call EMA over the CPU budget, expandFrontier -// should multiplicatively decrease the batch; under the budget it should additively increase toward -// the cap. -TEST_F(ReferenceChainsBfsTest, AdaptiveBatchSizeProportionalToWindow) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // Adaptive-batch state is zeroed by reset() (SetUp) but zeroed here too for the same reason as - // before: exact per-phase arithmetic below. - ReferenceChainsTestAccessor::setGotwEmaCallNs(0); - ReferenceChainsTestAccessor::setGotwBatchSize(0); - ReferenceChainsTestAccessor::setPassDeadlineNs(0); - - // Seed a frontier root manually (mirrors PollWatchedTargetsTest's seeding style): node carries - // frontier tag 1, pending expansion has exactly that tag. - int rootNode = addNode(); - int childNode = addNode(); - node_tags[rootNode] = 1; - ASSERT_TRUE(tracker->frontierTable()->insert( - 1, 0, 1, 0, FrontierEntryState::EDGE)); - ReferenceChainsTestAccessor::pushPendingExpandForTest(1); - int edges = 0; - const u64 budget = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); - - // --- Populate phase: first GetObjectsWithTags call. The EMA should be non-zero afterwards, and - // the near-zero mock call time means the window (nominal budget, no deadline) fits ~unbounded - // many calls - the proportion scales the batch all the way to the cap. - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - EXPECT_NE(0u, ReferenceChainsTestAccessor::gotwEmaCallNs()) - << "per-call EMA should be populated after first GetObjectsWithTags"; - EXPECT_EQ(ReferenceChainsTestAccessor::gotwMaxBatch(), - ReferenceChainsTestAccessor::gotwBatchSize()) - << "near-free call should scale the batch to the cap"; - - // --- Shrink phase: EMA at 2x the window with no deadline -> batch halves (512 x 1 / 1.6 after - // the EMA update). - ReferenceChainsTestAccessor::setGotwEmaCallNs(budget * 2); - ReferenceChainsTestAccessor::setGotwBatchSize(512); - ReferenceChainsTestAccessor::pushPendingExpandForTest(1); - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - // EMA after the call: 2x window x 0.8 + mock elapsed/5 - slightly above 1.6x window, so the - // exact expectation is computed from the actual EMA the same way the control law does (window = - // nominal budget, no deadline): next = 512 x window / ema. - EXPECT_EQ((size_t)(512ULL * budget / - std::max(ReferenceChainsTestAccessor::gotwEmaCallNs(), - 1ULL)), - ReferenceChainsTestAccessor::gotwBatchSize()) - << "EMA at ~1.6x the window should scale the batch to 512/1.6"; - - // --- Grow phase: EMA at half the window -> batch scales up 2.5x, i.e. the floor-dominated - // regime GROWS the batch (the whole point of the proportional law - the old AIMD could not grow - // past a fixed budget even when bigger batches were nearly free). - ReferenceChainsTestAccessor::setGotwBatchSize(64); - ReferenceChainsTestAccessor::setGotwEmaCallNs(budget / 2); - ReferenceChainsTestAccessor::pushPendingExpandForTest(1); - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - // Same computation from the actual post-call EMA (~0.4x window): next = 64 x window / ema. - EXPECT_EQ((size_t)(64ULL * budget / - std::max(ReferenceChainsTestAccessor::gotwEmaCallNs(), - 1ULL)), - ReferenceChainsTestAccessor::gotwBatchSize()) - << "EMA under the window should scale the batch up proportionally"; - - // --- Deadline-window phase: with a live pass deadline the window is the REMAINING time, not - // the nominal budget. - ReferenceChainsTestAccessor::setPassDeadlineNs( - OS::nanotime() + budget * 10); - ReferenceChainsTestAccessor::setGotwBatchSize(64); - ReferenceChainsTestAccessor::setGotwEmaCallNs(budget); - ReferenceChainsTestAccessor::pushPendingExpandForTest(1); - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - EXPECT_EQ(ReferenceChainsTestAccessor::gotwMaxBatch(), - ReferenceChainsTestAccessor::gotwBatchSize()) - << "a wide remaining deadline should grow the batch to the cap"; - ReferenceChainsTestAccessor::setPassDeadlineNs(0); - - // --- Admission sanity: expansion still walks the graph. Root -> child edge, one more drive, - // child must be admitted. - script.push_back({JVMTI_HEAP_REFERENCE_FIELD, rootNode, childNode, -1}); - ReferenceChainsTestAccessor::pushPendingExpandForTest(1); - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - EXPECT_NE(0, tags_ever_assigned[childNode]) - << "expandFrontier failed to admit childNode with adaptive batch_size"; - - tracker->stop(); -} - -// gotwWindowNs() backlog-pressure widening, unit level: the pod regime is a remaining pass window -// (~10ms) smaller than the measured per-call floor (~22-40ms at a 242k-entry tag map), against a -// lane 127k deep. -TEST_F(ReferenceChainsBfsTest, GotwWindowWidensOnlyUnderBacklogPressure) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - const u64 budget = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); - const size_t depth = ReferenceChainsTestAccessor::gotwBacklogMinDepth(); - const u64 mult = ReferenceChainsTestAccessor::gotwBacklogWindowMult(); - const u64 floor = budget * 2; // any floor above the nominal window - - // No deadline and no EMA yet: the nominal budget window. - ReferenceChainsTestAccessor::setGotwEmaCallNs(0); - EXPECT_EQ(budget, ReferenceChainsTestAccessor::gotwWindowNs(0, depth)); - - ReferenceChainsTestAccessor::setGotwEmaCallNs(floor); - - // Floor above the remaining window but a SHALLOW lane: no widening - the remaining window - // stands (rotation fast-lane stays cheap). - EXPECT_EQ(1u, ReferenceChainsTestAccessor::gotwWindowNs(1, 1)); - - // Floor above the remaining window and a DEEP lane: widened to EMA x mult, never below the - // remaining window itself. - EXPECT_EQ(floor * mult, - ReferenceChainsTestAccessor::gotwWindowNs(1, depth)); - - // Floor BELOW the remaining window: no widening even at depth - the ordinary proportional law - // already fits the call in the window. - ReferenceChainsTestAccessor::setGotwEmaCallNs(budget / 2); - EXPECT_EQ(budget, - ReferenceChainsTestAccessor::gotwWindowNs(budget, depth)); - ReferenceChainsTestAccessor::setGotwEmaCallNs(floor); - - // Floor above the NOMINAL window (deadline already passed, the exact pod's post-call state) at - // depth: still widened - the floor is paid by the next call regardless, so the batch must - // amortize it. - EXPECT_EQ(floor * mult, - ReferenceChainsTestAccessor::gotwWindowNs(0, depth)); - - tracker->stop(); -} - -// The widened window in action through the real control loop: one GetObjectsWithTags call whose -// floor (simulated by the mock's busy-wait) exceeds both the remaining pass deadline and the -// nominal window, with a backlog deeper than GOTW_BACKLOG_MIN_DEPTH, must GROW the calibrated batch -// (calib x mult, exactly - the window scales with the measured EMA) instead of collapsing it to -// GOTW_MIN_BATCH. -TEST_F(ReferenceChainsBfsTest, AdaptiveBatchGrowsWhenFloorExceedsWindowUnderDeepBacklog) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // A pass deadline a fraction of the simulated per-call floor: the call overruns it (exactly the - // pod's 10ms window vs 22-40ms floor), so after the call the loop's deadline check stops the - // invocation with ONE control update - deterministic arithmetic for the assertion below. - gotw_delay_ns = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); // 25ms floor - ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 5000000ULL); - ReferenceChainsTestAccessor::setGotwBatchSize(ReferenceChainsTestAccessor::gotwMinBatch()); - ReferenceChainsTestAccessor::setGotwEmaCallNs(0); // seeded by the call below - - // A pending lane deep enough to cross GOTW_BACKLOG_MIN_DEPTH. - const size_t depth = ReferenceChainsTestAccessor::gotwBacklogMinDepth() + 1; - for (size_t i = 0; i < depth; i++) { - ReferenceChainsTestAccessor::pushPendingExpandForTest( - (jlong)(1000000 + i)); - } - - int edges = 0; - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - - // EMA after the call = the busy-wait floor (~25ms). The pass deadline is long past, so the - // window is widened to EMA x GOTW_BACKLOG_WINDOW_MULT, and the control law computes calib x - // window / ema = calib x mult - exactly, because the window is a whole multiple of the same EMA - // it divides by. - EXPECT_EQ(ReferenceChainsTestAccessor::gotwMinBatch() * - ReferenceChainsTestAccessor::gotwBacklogWindowMult(), - ReferenceChainsTestAccessor::gotwBatchSize()) - << "the floor-dominated deep-backlog regime must GROW the batch, " - "not clamp it to GOTW_MIN_BATCH"; - - ReferenceChainsTestAccessor::setPassDeadlineNs(0); - gotw_delay_ns = 0; - tracker->stop(); -} - -// FAIR-SHARE DRAIN persistence: the lane toggle must survive across expandFrontier() invocations. -TEST_F(ReferenceChainsBfsTest, FairShareLaneAlternationPersistsAcrossInvocations) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int rootNode = addNode(); - int otherRoot = addNode(); - // Two live boundary objects: tag 1 in pending, tag 2 in priority. - node_tags[rootNode] = 1; - node_tags[otherRoot] = 2; - ASSERT_TRUE(tracker->frontierTable()->insert( - 1, 0, 1, 0, FrontierEntryState::EDGE)); - ASSERT_TRUE(tracker->frontierTable()->insert( - 2, 0, 1, 0, FrontierEntryState::EDGE)); - ReferenceChainsTestAccessor::pushPendingExpandForTest(1); - ReferenceChainsTestAccessor::pushPriorityExpand(2); - int edges = 0; - - // Mock GetObjectsWithTags calls are ~free, so without a deadline a single expandFrontier() - // invocation would drain BOTH lanes in one loop. - gotw_delay_ns = 1 * 1000 * 1000; // 1ms - ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 200 * 1000); - - // Invocation 1: priority first (the standing preference). - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - EXPECT_EQ(0u, ReferenceChainsTestAccessor::priorityExpandSize()) - << "first invocation should drain the priority lane"; - EXPECT_EQ(1u, ReferenceChainsTestAccessor::pendingExpandSize()) - << "first invocation must leave the pending lane for the next one"; - EXPECT_FALSE(ReferenceChainsTestAccessor::expandLanePreferPriority()); - - // Rotation refills the priority lane; invocation 2 must STILL prefer the pending lane - the - // toggle persists, it is not reset per call. - ReferenceChainsTestAccessor::pushPriorityExpand(2); - ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 200 * 1000); - ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, - &mock_jni, &edges); - EXPECT_EQ(0u, ReferenceChainsTestAccessor::pendingExpandSize()) - << "second invocation should drain the pending lane"; - EXPECT_EQ(1u, ReferenceChainsTestAccessor::priorityExpandSize()) - << "second invocation must leave the refilled priority lane alone"; - EXPECT_TRUE(ReferenceChainsTestAccessor::expandLanePreferPriority()); - - tracker->stop(); -} -// FANOUT HYGIENE: a _leak_parent_fanout entry whose parent no longer resolves in the frontier -// (pruned: dead object, or a search-restart wipe) can never be re-walked, so -// collectStaleExpandedEntriesForRotation() must erase it during selection rather than skip it -// forever - without the erase, the fanout grows monotonically with corpses (observed live at ~11k -// entries of overwhelmingly-dead old backing arrays), which both bloats the selection scan and -// turns the fanout cursor's lap arithmetic into mostly wasted skips. -TEST_F(ReferenceChainsBfsTest, StaleRotationEvictsDeadFanoutParents) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - constexpr u32 kLeafKlass = 987; - ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); - FrontierTable *frontier = tracker->frontierTable(); - - // Live fanout parent 1 and dead fanout parent 5 (frontier entry pruned). - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 1, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, - /*class_tag=*/42)); - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, 5, 0, 0, FrontierEntryState::EXPANDED, - JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, - /*class_tag=*/43)); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 1, 10); - ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 5, 20); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(1)); - ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(5)); - - frontier->clear(5); // parent 5's object died / search restart pruned it - - std::vector selected = - ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(4); - ASSERT_EQ(1u, selected.size()); - EXPECT_EQ((jlong)1, selected[0]); - EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakParentFanout(5)) - << "dead fanout parent must be erased during selection"; - EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(1)) - << "live fanout parent must survive"; - - tracker->stop(); -} - - -// Leak-tag interception (design A + C): an object pre-tagged with a leak tag (as -// LivenessTracker::tagLeakInstances() would have set on a tracked leaking instance) must be -// admitted by converting the leak tag to a frontier tag, with the leak tag preserved in the -// frontier entry so buildChainEvent() emits it as target_tag - the ReferenceChain <-> -// HeapLiveObject correlation key. -TEST_F(ReferenceChainsBfsTest, LeakTagInterceptionConvertsToFrontierTagAndCorrelates) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); - - int rootNode = addNode(); - int leakChild = addNode(); - int plainChild = addNode(); - // Simulate tagLeakInstances(): the tracked leaking instance already carries a leak tag; the - // sibling does not. - node_tags[leakChild] = leak_tag; - script = { - {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, rootNode, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, rootNode, leakChild, -1}, - {JVMTI_HEAP_REFERENCE_FIELD, rootNode, plainChild, -1}, - }; - - // A single pass drains this tiny graph to completion, and a completed search releases all JVMTI - // tags (releaseSearchTags(), "tagsReleased" in runPass's own log) - so read the tags from - // tags_ever_assigned, which records each tag at assignment time and is never reset (see its own - // comment), not from node_tags (which reads 0 after release). - bool truncated = true; - ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); - - // The leak-tagged child's tag was REPLACED by a frontier tag (small positive, outside the leak - // range). - jlong leak_ftag = tags_ever_assigned[leakChild]; - ASSERT_NE(leak_tag, leak_ftag) - << "leak tag was never intercepted - BFS did not reach the object"; - ASSERT_GT(leak_ftag, 0); - EXPECT_LT(leak_ftag, leak_tag) << "frontier tag must be outside leak range"; - - // The frontier entry preserves the leak tag for correlation. - EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(leak_ftag)); - - // The untagged sibling got an ordinary admit: frontier tag assigned, but no leak tag stored. - jlong plain_ftag = tags_ever_assigned[plainChild]; - ASSERT_GT(plain_ftag, 0); - EXPECT_EQ(0, ReferenceChainsTestAccessor::frontierLeakTag(plain_ftag)); - - // Design C: buildChainEvent() reports the leak tag as target_tag for the leak-tagged instance, - // and the plain frontier tag for the sibling. - ReferenceChainEvent event; - ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( - &mock_jvmti, &mock_jni, leak_ftag, &event)); - EXPECT_EQ((u64)leak_tag, event._target_tag) - << "chain target tag must be the leak tag (correlation key)"; - EXPECT_GE(event._depth, 1u) << "leak child sits behind the root, not at it"; - - ReferenceChainEvent plain_event; - ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( - &mock_jvmti, &mock_jni, plain_ftag, &plain_event)); - EXPECT_EQ((u64)plain_ftag, plain_event._target_tag) - << "untagged instance must keep the frontier tag as target tag"; - - tracker->stop(); -} - -// Candidate-scoped reach, prong 1 (walkCandidateThreadLocals()): a leak held through the leaking -// thread's ThreadLocalMap must be intercepted with its full chain by ONE bounded walk from the -// Thread object, no matter what the ordinary BFS backlog state is - and the walk's gates must keep -// it off the Thread's non-thread-local fields entirely. -TEST_F(ReferenceChainsBfsTest, ThreadWalkDescendsOnlyThreadLocalMapAndInterceptsLeak) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - // The classes descendFromAnchor()'s resolutions look up by FindClass name (see - // registerClassForFindClass' own comment) + the scripted graph's own classes. - void *threadCls = (void *)0x3001, *tlmapCls = (void *)0x3002, - *loaderCls = (void *)0x3003, *holderCls = (void *)0x3004, - *chunkCls = (void *)0x3005; - int tlmapIdx = - registerClassForFindClass(tlmapCls, - "java/lang/ThreadLocal$ThreadLocalMap", - "Ljava/lang/ThreadLocal$ThreadLocalMap;"); - int loaderIdx = - registerClassForFindClass(loaderCls, "java/lang/ClassLoader", - "Ljava/lang/ClassLoader;"); - registerClassForFindClass(threadCls, "java/lang/Thread", - "Ljava/lang/Thread;"); - int holder = addClass(holderCls, "Lcom/rc/descendwalk/Holder;"); - int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/LeakChunk;"); - thread_class = threadCls; - ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); - - int threadNode = addNode(); - int threadNode2 = addNode(); // second candidate thread: fresh-admission path - int tlmapNode = addNode(); - int loaderNode = addNode(); // Thread's contextClassLoader: anchor gate - int loaderNode2 = addNode(); // a ClassLoader below the gate: no-descend - int holderNode = addNode(); - int leakChunk = addNode(); - - const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); - node_tags[leakChunk] = leak_tag; - - // Topological order (mock_FollowReferences replays edges in script order, expanding only refs - // the production callback said to descend into). - script = { - {JVMTI_HEAP_REFERENCE_FIELD, threadNode, tlmapNode, - /*class_idx=*/-1}, - {JVMTI_HEAP_REFERENCE_FIELD, threadNode, loaderNode, - /*class_idx=*/-1}, - {JVMTI_HEAP_REFERENCE_FIELD, tlmapNode, holderNode, holder}, - {JVMTI_HEAP_REFERENCE_FIELD, tlmapNode, loaderNode2, - /*class_idx=*/-1}, - {JVMTI_HEAP_REFERENCE_FIELD, holderNode, leakChunk, chunk}, - }; - // The anchor gate compares the REFEREE's class tag, so the thread edges' class_idx values - // matter: the tlmap edge carries ThreadLocalMap's tag, and the loader edges ClassLoader's. - script[0].class_idx = tlmapIdx; - script[1].class_idx = loaderIdx; - script[3].class_idx = loaderIdx; - - // The first thread walks the REUSE path: its Thread object is already admitted (root-attached - // THREAD entry + JVMTI tag) exactly as it is in production after the first walk pass or root - // enumeration. - FrontierTable *frontier = tracker->frontierTable(); - jlong anchor_tag = - tracker->tagObject(&mock_jvmti, - reinterpret_cast(&node_tags[threadNode])); - ASSERT_GT(anchor_tag, 0); - node_tags[threadNode] = anchor_tag; - ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( - frontier, anchor_tag, 0, 0, FrontierEntryState::FRONTIER, - (u8)JVMTI_HEAP_REFERENCE_THREAD)); - - // Two candidate slots, two qualifying tids: tid 777's thread is pre-anchored (reuse path), tid - // 778's is untagged (fresh-admission path - GetObjectClass + tagObject + insert, the path a - // never-walked thread takes in production). - jint tids0[] = {777}; - jint tids1[] = {778}; - ReferenceChainsTestAccessor::seedCandidateSlotForTest( - /*slot=*/0, /*klass_id=*/6, tids0, 1); - ReferenceChainsTestAccessor::seedCandidateSlotForTest( - /*slot=*/1, /*klass_id=*/6, tids1, 1); - tracker->registerThreadObject( - &mock_jni, 777, reinterpret_cast(&node_tags[threadNode])); - tracker->registerThreadObject( - &mock_jni, 778, reinterpret_cast(&node_tags[threadNode2])); - - int edges = 0; - ReferenceChainsTestAccessor::walkCandidateThreadLocalsForTest( - &mock_jvmti, &mock_jni, 1000, &edges); - - // The ThreadLocalMap-held chain was admitted end-to-end and the leak-tagged chunk was - // intercepted (tag replaced by a frontier tag, leak tag preserved for correlation). - jlong thread_ftag = anchor_tag; - jlong tlmap_ftag = tags_ever_assigned[tlmapNode]; - ASSERT_GT(tlmap_ftag, 0) << "anchor gate did not descend into ThreadLocalMap"; - jlong holder_ftag = tags_ever_assigned[holderNode]; - ASSERT_GT(holder_ftag, 0) << "walk did not descend below ThreadLocalMap"; - jlong chunk_ftag = tags_ever_assigned[leakChunk]; - ASSERT_NE(chunk_ftag, leak_tag) - << "leak-tagged chunk under the ThreadLocalMap was never intercepted"; - ASSERT_GT(chunk_ftag, 0); - EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); - - // Chain shape: Thread (root-attached, THREAD root kind) -> ThreadLocalMap -> holder -> chunk. - FrontierEntry thread_entry{}; - ASSERT_TRUE(frontier->lookup(thread_ftag, &thread_entry)); - EXPECT_EQ(0, thread_entry.parent_tag); - EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_THREAD, thread_entry.root_kind); - FrontierEntry chunk_entry{}; - ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); - EXPECT_EQ(holder_ftag, chunk_entry.parent_tag); - EXPECT_EQ(3u, chunk_entry.depth); - - // The gates kept the walk off the metadata branches: neither the Thread's own - // contextClassLoader edge (anchor gate) nor a ClassLoader below ThreadLocalMap (no-descend - // gate) was admitted. - EXPECT_EQ(0, tags_ever_assigned[loaderNode]) - << "anchor gate must not admit the Thread's non-ThreadLocalMap fields"; - EXPECT_EQ(0, tags_ever_assigned[loaderNode2]) - << "no-descend gate must not admit fat-metadata classes below the anchor"; - - // The second thread took the fresh-admission path (no prior tag/entry): its Thread object was - // admitted root-attached with the THREAD root kind. - jlong thread2_ftag = ReferenceChainsTestAccessor::getTagForTest( - &mock_jvmti, reinterpret_cast(&node_tags[threadNode2])); - ASSERT_GT(thread2_ftag, 0) << "fresh thread anchor was never admitted"; - FrontierEntry thread2_entry{}; - ASSERT_TRUE(frontier->lookup(thread2_ftag, &thread2_entry)); - EXPECT_EQ(0, thread2_entry.parent_tag); - EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_THREAD, thread2_entry.root_kind); - - tracker->stop(); -} - -// unregisterThreadObject() must defer the global-ref deletion to releaseEndedThreadRefs(): -// walkCandidateThreadLocals() copies the jobject out of _thread_objects under _thread_objects_lock, -// releases the lock, and can still be using it as a FollowReferences anchor when a concurrent -// ThreadEnd erases the entry - deleting there would be JNI use-after-free (see -// _thread_refs_pending_delete's comment). -TEST_F(ReferenceChainsBfsTest, ThreadRefUnregisterDefersGlobalRefDeletion) { - Arguments args; - ASSERT_FALSE(args.parse("referencechains=true")); - ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); - ASSERT_FALSE(tracker->start(args)); - - int threadNode = addNode(); - tracker->registerThreadObject( - &mock_jni, 555, reinterpret_cast(&node_tags[threadNode])); - tracker->unregisterThreadObject(&mock_jni, 555); - // The erasing side only enqueues - no DeleteGlobalRef yet. - EXPECT_EQ(0, global_refs_deleted_); - - // The drain deletes exactly the queued ref, and draining an empty list is a no-op. - tracker->releaseEndedThreadRefs(&mock_jni); - EXPECT_EQ(1, global_refs_deleted_); - tracker->releaseEndedThreadRefs(&mock_jni); - EXPECT_EQ(1, global_refs_deleted_); - - tracker->stop(); -} - -// Candidate-scoped reach, prong 2 (collectStaticFieldAnchorsForRotation()/ -// walkStaticFieldAnchors()): the collector selects exactly the root-attached static-holder entries -// with a wrapping cursor, and the walk reaches a leak held 3-4 hops inside a static collection in -// one bounded call - the shape the one-hop Tier-2 rotation demonstrably cannot reach from an -// un-expanded FRONTIER holder on a rising heap (pod rounds 5-6). diff --git a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp index a64e28382e..64aa4c53c5 100644 --- a/ddprof-lib/src/test/cpp/referenceChains_ut.cpp +++ b/ddprof-lib/src/test/cpp/referenceChains_ut.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -40,15 +41,7404 @@ class ReferenceChainsGlobalSetup { static ReferenceChainsGlobalSetup global_setup; -// Enable diagnostics for tests. +// The rc debug-log gate defaults to silent (0) so DEBUG builds are +// pod-safe (see rcDebugLevel.h). Tests want the full diagnostics that +// used to be unconditional: pin level 2 before any library code runs. [[maybe_unused]] static const bool kRcDebugLevelPinnedForTests = setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1) == 0; -#include "referenceChainsCoreTests.inc" -#include "referenceChainsBfsTests.inc" -#include "referenceChainsPodTests.inc" -#include "referenceChainsTrackerTests.inc" -#include "referenceChainsRotationTests.inc" -#include "referenceChainsTraversalTests.inc" -#include "referenceChainsAnchorTests.inc" -#include "referenceChainsEventTests.inc" +// --------------------------------------------------------------------------- +// RcDebugLevelTest - the runtime debug-log gate's pure parsing + env +// plumbing (rcDebugLevel.h). No tracker state involved. +// --------------------------------------------------------------------------- +TEST(RcDebugLevelTest, ParseAcceptsTrimmedSingleDigits) { + EXPECT_EQ(parseRcDebugLevel(nullptr), -1); + EXPECT_EQ(parseRcDebugLevel(""), -1); + EXPECT_EQ(parseRcDebugLevel("0"), 0); + EXPECT_EQ(parseRcDebugLevel("1"), 1); + EXPECT_EQ(parseRcDebugLevel("2"), 2); + EXPECT_EQ(parseRcDebugLevel(" 2\n"), 2); // the `echo 2 >` form + EXPECT_EQ(parseRcDebugLevel("\t1\r\n"), 1); + EXPECT_EQ(parseRcDebugLevel("3"), -1); + EXPECT_EQ(parseRcDebugLevel("12"), -1); + EXPECT_EQ(parseRcDebugLevel("-1"), -1); + EXPECT_EQ(parseRcDebugLevel("abc"), -1); + EXPECT_EQ(parseRcDebugLevel("2 garbage"), -1); +} + +TEST(RcDebugLevelTest, ReadFileParsesTrimmedAndRejectsInvalid) { + char path[] = "/tmp/rc_dbg_test_XXXXXX"; + int fd = mkstemp(path); + ASSERT_GE(fd, 0); + close(fd); + struct Case { + const char *content; + int expected; + }; + const Case cases[] = { + {"2\n", 2}, {"1", 1}, {" 2 ", 2}, {"\n1\n", 1}, + {"", -1}, {"3", -1}, {"22", -1}, {"abc", -1}, {"x", -1}, + }; + for (const Case &c : cases) { + FILE *f = fopen(path, "w"); + ASSERT_NE(f, nullptr); + EXPECT_GE(fputs(c.content, f), 0); // non-negative on success + fclose(f); + EXPECT_EQ(readRcDebugLevelFile(path), c.expected) << "content='" << c.content << "'"; + } + unlink(path); + EXPECT_EQ(readRcDebugLevelFile(path), -1); // now missing + EXPECT_EQ(readRcDebugLevelFile(nullptr), -1); +} + +TEST(RcDebugLevelTest, RefreshFollowsEnvWhenNoOverrideFile) { + // The refresh's fixed override path is machine-global; skip rather + // than flake on a developer machine that happens to have the file. + if (access("/tmp/ddprof_root/refchains_debug_level", F_OK) == 0) { + GTEST_SKIP() << "override file present on this machine"; + } + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "1", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 1); + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 2); + // invalid env value means silent, not "keep the previous level" + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "bogus", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 0); + // restore the pinned level for any later test's diagnostics + setenv("DD_PROFILING_REFERENCE_CHAINS_DEBUG", "2", 1); + rcDebugLevelRefresh(true); + EXPECT_EQ(rcDebugLevel(), 2); +} + +// --------------------------------------------------------------------------- +// VMTestAccessor - friend of VM (vmEntry.h), lets tests swap VM::_jvmti for a +// mock. This gtest binary has no live JVM attached (see jvmSupport_ut.cpp's +// fixture comment for the same constraint on a different subsystem), but +// ReferenceChainTracker::start()/stop() now call VM::jvmti()-> +// SetEventNotificationMode() (the lazy event-enable step), so a mock is +// required for those calls to be exercised without crashing on a null +// jvmtiEnv. +// --------------------------------------------------------------------------- +class VMTestAccessor { +public: + static jvmtiEnv* getJvmti() { return VM::_jvmti; } + static void setJvmti(jvmtiEnv* env) { VM::_jvmti = env; } +}; + +// --------------------------------------------------------------------------- +// ReferenceChainsTestAccessor - same pattern as VMTestAccessor above, for the +// same reason: ReferenceChainTracker::instance() is a process-wide singleton +// (referenceChains.h), so the search-lifecycle fields +// (_search_state/_search_started/_pending_expand/...) would otherwise leak +// from one ReferenceChainsBfsTest TEST_F into the next in this same gtest +// binary - e.g. a test that drives the search to SearchState::COMPLETED +// would leave every later test's runPass() call a permanent no-op (see +// runPass()'s "already terminal -> no-op" branch). reset() puts the tracker +// back to its just-constructed state; it does not change production +// behavior. +// --------------------------------------------------------------------------- +class ReferenceChainsTestAccessor { +public: + static void reset() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + delete t->_frontier; + t->_frontier = nullptr; + t->_class_tags = ClassTagTable(); + t->_last_resolved_class_count = 0; + // Without these two, a prior test's fully-swept (or + // partially-swept) admitStaticFieldRoots() state survives in this + // process-wide singleton and can wrongly skip the sweep entirely on + // this test's first pass if its resolved class count happens to + // match whatever an earlier test last left behind - see + // resetForRestart()'s identical reset of these same fields for the + // production-restart equivalent of this same contract. + t->_last_static_field_class_count = -1; + t->_static_field_sweep_cursor = 0; + t->_static_field_sweep_cycle_truncated = false; + t->_next_tag = 1; + // Shared with LivenessTracker (classTagAllocator.h) - process-wide, + // not per-ReferenceChainTracker-instance, so it needs its own reset + // seam rather than being a plain member write. + ClassTagAllocator::resetForTest(); + t->_search_started = false; + t->_tags_released = true; + t->_search_state = SearchState::RUNNING; + t->_abandon_reason = SearchAbandonReason::NONE; + t->_search_start_ns = 0; + t->_pending_expand.clear(); + t->_priority_expand.clear(); + t->_priority_expand_set.clear(); + // B' at-risk FIFO + its indexes/counters, and the round-15 fresh + // lane: production clears all of these on every restartSearch()/ + // resetSearchStateForTest(), but this seam predates the FIFO and was + // never given the clears - until round 16 a test left at-risk + // entries behind and only survived because LATER tests' own + // runPass()s drained the residue with the full Bfs mock. A test + // that ends with a saturated FIFO (AtRiskFifoPerClassQuotaKeeps- + // FloodOut) exposed the gap: the residue drained into the NEXT + // fixture's runPass, whose minimal mock has a null + // GetObjectsWithTags (walkStaticFieldAnchors crash). The seam's + // contract is "empty process-wide singleton state for the next + // test" - state everything, assert nothing about prior residue. + t->_static_anchor_fifo.clear(); + t->_static_anchor_fifo_set.clear(); + t->_static_anchor_fifo_klass_counts.clear(); + t->_static_anchor_fresh_queue.clear(); + t->_last_pass_gc_finish_epoch = 0; + t->_last_pass_ns = 0; + t->_passes_run = 0; + t->_passes_since_last_progress = 0; + t->_candidate_count = 0; + t->_candidate_found_bits = 0; + memset(t->_candidate_discovered_count, 0, sizeof(t->_candidate_discovered_count)); + t->_passes_since_last_candidate_progress = 0; + t->_last_candidate_progress_mark = 0; + t->_canary_stuck_restart_count = 0; + t->_resolved_chains.clear(); + t->_safepoint_pain_budget = PainBudget(); + t->_cpu_pain_budget = PainBudget(); + t->_search_pain_ms = 0; + t->_root_kind_rotation_cursor = 1; + t->_stale_expanded_rotation_cursor = 1; + t->_static_anchor_index.clear(); + t->_static_anchor_own_class_tags.clear(); + t->_static_anchor_index_tags.clear(); + t->_anchor_container_cursor = 0; + t->_anchor_other_cursor = 0; + t->_class_shape_cache.clear(); + t->_thread_walk_anchor_cursor = 0; + memset(t->_candidate_qualifying_tid_count, 0, + sizeof(t->_candidate_qualifying_tid_count)); + t->_hop_label_cache.clear(); + t->_watched_leak_klass_count = 0; + t->_leak_signature_totals.clear(); + t->_leak_signature_prev_totals.clear(); + t->_leak_parent_fanout.clear(); + t->_borrowed_budget = 0; + t->_consecutive_under_target_passes = 0; + // Adaptive batch + lane state: NOT covered by anything above, and a + // prior test that drove expansion leaves a non-zero EMA, a live + // batch size, a stale pass deadline, and/or a mid-alternation lane + // toggle behind - all of which silently change the next test's + // expandFrontier() arithmetic (exact-value asserts on batch sizing + // only pass standalone otherwise). + t->_gotw_ema_call_ns = 0; + t->_gotw_batch_size = 0; + t->_pass_deadline_ns = 0; + t->_expand_lane_prefer_priority = true; + } + + // Search restart + pain budget (SearchRestartTest below) - same + // rationale as the pacing accessors above: private state a test needs to + // drive/observe directly. + static bool canAffordNewSearch(u64 now_ns) { + return ReferenceChainTracker::instance()->canAffordNewSearch(now_ns); + } + + static bool shouldRunPass(u64 now_ns) { + return ReferenceChainTracker::instance()->shouldRunPass(now_ns); + } + + static void setSearchPainMs(u64 ms) { + ReferenceChainTracker::instance()->_search_pain_ms = ms; + } + + static void setCandidateFrontierTagForTest(int idx, jlong tag) { + ReferenceChainTracker::instance()->setCandidateFrontierTagForTest(idx, tag); + } + + static void setCandidateCountForTest(int n) { + ReferenceChainTracker::instance()->setCandidateCountForTest(n); + } + + // Canary-lane backoff state wrappers - see _canary_backoff_mult's own + // comment (referenceChains.h). + static int canaryBackoffMult() { + return ReferenceChainTracker::instance()->canaryBackoffMultForTest(); + } + static u64 lastCanaryPassNs() { + return ReferenceChainTracker::instance()->lastCanaryPassNsForTest(); + } + static void setCanaryBackoffForTest(int mult, u64 ema_ms, u64 last_pass_ns) { + ReferenceChainTracker::instance()->setCanaryBackoffForTest(mult, ema_ms, + last_pass_ns); + } + static void setOomRampActive(bool active) { + ReferenceChainTracker::instance()->setOomRampActiveForTest(active); + } + + static int passesSinceLastCandidateProgress() { + return ReferenceChainTracker::instance()->passesSinceLastCandidateProgressForTest(); + } + + static int canaryStuckRestartCount() { + return ReferenceChainTracker::instance()->canaryStuckRestartCountForTest(); + } + + static u64 searchPainMs() { + return ReferenceChainTracker::instance()->_search_pain_ms; + } + + // Resolved-chain cache: read-only size peek and a pass-through to the + // private snapshot (drainPendingChainEvents()) and insert + // (cacheResolvedChain()), for ResolvedChainCacheTest below - same + // rationale as hasResolvedChainForTag()/resolvedChainCount() below. + static size_t resolvedChainCount() { + return ReferenceChainTracker::instance()->_resolved_chains.size(); + } + + static void drain(std::vector *out) { + ReferenceChainTracker::instance()->drainPendingChainEvents(out); + } + + static void cacheChain(jlong source_tag, ReferenceChainEvent event, + u64 source_search_ns) { + ReferenceChainTracker::instance()->cacheResolvedChain( + source_tag, std::move(event), source_search_ns); + } + + static int maxResolvedChains() { + return ReferenceChainTracker::MAX_RESOLVED_CHAINS; + } + + // Target-selection bridging step: read-only peeks into the resolved-chain + // cache, for asserting exactly which klass a chain was resolved for and + // the tag it was reconstructed from - see PollWatchedTargetsTest below. + static bool hasResolvedChainForTag(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return t->_resolved_chains.find(tag) != t->_resolved_chains.end(); + } + + static jlong resolvedChainSourceTag(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + auto it = t->_resolved_chains.find(tag); + return it == t->_resolved_chains.end() ? 0 : it->second.source_tag; + } + + // Leak-tag correlation (design C): read a frontier entry's stored leak + // tag, and a pass-through to the private buildChainEvent(), for + // LeakTagInterceptionTest below - same friend-accessor rationale as + // hasResolvedChainForTag() above. Returns -1 when the tag is not in the + // frontier table. + static jlong frontierLeakTag(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + FrontierEntry entry{}; + if (t->_frontier == nullptr || !t->_frontier->lookup(tag, &entry)) { + return -1; + } + return entry.leak_tag; + } + + static void setCandidateKlassIdForTest(int idx, u32 klass_id) { + ReferenceChainTracker::instance()->setCandidateKlassIdForTest(idx, klass_id); + } + + // Round-19: the leak-tag canary found criterion (pod 289f8 — the + // chase was structurally unresolvable after the marker->leak-tag + // migration; see buildDiscoveredInstanceChains' own comment). + static u64 candidateFoundBitsForTest() { + return ReferenceChainTracker::instance()->_candidate_found_bits; + } + + static jlong candidateFrontierTagForTest(int slot) { + return ReferenceChainTracker::instance()->_candidate_frontier_tags[slot]; + } + + static void buildDiscoveredInstanceChainsForTest(u32 klass_id, + u64 current_search_ns) { + // jvmti/jni null is safe: resolveHopEdgeLabel() null-guards and + // degrades hop labels to kind labels. + ReferenceChainTracker::instance()->buildDiscoveredInstanceChains( + nullptr, nullptr, klass_id, current_search_ns); + } + + // ---- pod-in-a-jar system harness (design-pod-in-a-jar-harness) ---- + static u8 searchStateForTest() { + return ReferenceChainTracker::instance()->_search_state; + } + + static int sweepGateResolvedCountForTest() { + return ReferenceChainTracker::instance()->_last_resolved_class_count; + } + + static int sweepGateStaticCountForTest() { + return ReferenceChainTracker::instance()->_last_static_field_class_count; + } + + static int sweepCursorForTest() { + return ReferenceChainTracker::instance()->_static_field_sweep_cursor; + } + + static int passesRunForTest() { + return ReferenceChainTracker::instance()->_passes_run; + } + + static int candidateCountForTest() { + return ReferenceChainTracker::instance()->_candidate_count; + } + + static u32 candidateKlassIdForTest(int slot) { + return ReferenceChainTracker::instance()->_candidate_klass_ids[slot]; + } + + static size_t resolvedChainCountForTest() { + return ReferenceChainTracker::instance()->_resolved_chains.size(); + } + + static std::vector resolvedChainTargetsForTest() { + std::vector out; + for (auto &kv : + ReferenceChainTracker::instance()->_resolved_chains) { + out.push_back(kv.second.event._target_tag); + } + return out; + } + + static size_t staticAnchorFreshQueueSizeForTest() { + return ReferenceChainTracker::instance() + ->_static_anchor_fresh_queue.size(); + } + + static jlong candidateDiscoveredTagForTest(int slot, int idx) { + return ReferenceChainTracker::instance()->candidateDiscoveredTagForTest(slot, idx); + } + + static int candidateDiscoveredCountForTest(int slot) { + return ReferenceChainTracker::instance()->candidateDiscoveredCountForTest(slot); + } + + // recordDiscoveredInstance()/correlateAdmittedLeakTag() are the + // production paths for the leak-correlation tests below. + static void recordDiscoveredInstanceForTest(u32 klass_id, jlong tag, + bool leak_correlated) { + ReferenceChainTracker::instance()->recordDiscoveredInstance(klass_id, tag, + leak_correlated); + } + + // Drive restartSearch() directly (the accessor base already set + // _tags_released, so its assert is satisfied). + static void restartSearchForTest() { + ReferenceChainTracker::instance()->restartSearch(); + } + + static bool anchorIndexIsEmptyForTest() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return t->_static_anchor_index.empty() && + t->_static_anchor_own_class_tags.empty() && + t->_static_anchor_index_tags.empty(); + } + + // Read back discovered-instance slots (frontier tags recorded by + // recordDiscoveredInstance). + static jlong discoveredTagForTest(int slot, int idx) { + return ReferenceChainTracker::instance() + ->_candidate_discovered_tags[slot][idx]; + } + + static int discoveredCountForTest(int slot) { + return ReferenceChainTracker::instance() + ->_candidate_discovered_count[slot]; + } + + static size_t priorityExpandCap() { + return ReferenceChainTracker::PRIORITY_EXPAND_CAP; + } + + static int maxDiscoveredPerClass() { + return ReferenceChainTracker::MAX_DISCOVERED_INSTANCES_PER_CLASS; + } + + static bool buildChainEventForTest(jvmtiEnv *jvmti, JNIEnv *jni, + jlong tag, ReferenceChainEvent *out) { + return ReferenceChainTracker::instance()->buildChainEvent(jvmti, jni, + tag, out); + } + + // Direct expandFrontier() drive for the AIMD batch test: a full runPass() + // drains a small graph to completion and its rotation phase adds extra + // GetObjectsWithTags calls, so per-call AIMD assertions cannot be made + // deterministic through runPass(). Seeding _pending_expand and calling + // expandFrontier() directly runs exactly one batch (one AIMD update). + static void pushPendingExpandForTest(jlong tag) { + ReferenceChainTracker::instance()->_pending_expand.push_back(tag); + } + + static void expandFrontierForTest(jvmtiEnv *jvmti, JNIEnv *jni, + int *edges_admitted) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + bool truncated = false; + bool cap_hit = false; + u64 safepoint_ticks = 0; + t->expandFrontier(jvmti, jni, t->_hop_cap, 1000, edges_admitted, + &truncated, &cap_hit, &safepoint_ticks); + } + + // Pause-time pacing controller: read-only peeks at the controller's + // derived values, and + // a pass-through to the private updatePacing() itself, for + // ReferenceChainsPacingTest below - same rationale as + // hasResolvedChainForTag()/resolvedChainCount() above (the target-selection bridging step): private state a test needs to drive/ + // observe directly, exposed via this existing friend accessor rather + // than adding public getters/setters to ReferenceChainTracker itself. + static int effectiveBudget() { + return ReferenceChainTracker::instance()->_effective_budget; + } + + static u64 effectiveCadenceNs() { + return ReferenceChainTracker::instance()->_effective_cadence_ns; + } + + static void updatePacing(u64 pass_wall_ns) { + ReferenceChainTracker::instance()->updatePacing(pass_wall_ns); + } + + static u64 baselineCadenceNs() { return ReferenceChainTracker::PASS_CADENCE_NS; } + + // Test-only seams for PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling + // below, which needs to start from a controlled below-ceiling/above- + // baseline point with a freshly reset controller (see that test's own + // comment for why chaining directly off a prior constant-input sequence + // would leave _pause_pid's integral state mid-recovery from that + // sequence's windup, muddying this method's per-step direction + // assertions with a transient the test is not about). + static void setEffectiveBudget(int v) { + ReferenceChainTracker::instance()->_effective_budget = v; + } + + static void setEffectiveCadenceNs(u64 v) { + ReferenceChainTracker::instance()->_effective_cadence_ns = v; + } + + static void resetPacingController() { + ReferenceChainTracker::instance()->_pause_pid.reset(); + } + + // Budget-borrowing (referenceChains.h's _borrowed_budget comment): the + // configured multiplier PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling + // below asserts convergence against, instead of hardcoding it a second + // time in the test itself. + static int borrowCeilingMultiplier() { + return ReferenceChainTracker::BORROW_CEILING_MULTIPLIER; + } + + static int64_t borrowedBudget() { + return ReferenceChainTracker::instance()->_borrowed_budget; + } + + // MaybeRevokeBorrowForRootEnumPass* tests below: drive the borrow state + // directly into "already granted" before exercising the revocation-only + // seam, and a pass-through to that seam itself - same rationale as + // updatePacing()'s own accessor above. + static void setBorrowedBudget(int64_t v) { + ReferenceChainTracker::instance()->_borrowed_budget = v; + } + + static int consecutiveUnderTargetPasses() { + return ReferenceChainTracker::instance()->_consecutive_under_target_passes; + } + + static void setConsecutiveUnderTargetPasses(int v) { + ReferenceChainTracker::instance()->_consecutive_under_target_passes = v; + } + + static void maybeRevokeBorrowForRootEnumPass(u64 pass_wall_ticks) { + ReferenceChainTracker::instance()->maybeRevokeBorrowForRootEnumPass( + pass_wall_ticks); + } + + // ReleaseSearchTagsFailureTest below: read-only peek at whether the + // tracker still owes a tag release before it can allow a restart - see + // _tags_released's own comment. + static bool tagsReleased() { + return ReferenceChainTracker::instance()->_tags_released; + } + + // ResolveLoadedClassesRescansAfterClassCountShrinksAndPartiallyRegrows + // below: direct pass-through to the private resolveLoadedClasses(), plus + // a read-only peek at the count it stashes - the same rationale as + // tagsReleased() above (private state/behavior a test needs to + // drive/observe directly, without going through a full runPass()/search + // lifecycle that resolveLoadedClasses() alone does not need). + static void resolveLoadedClasses(jvmtiEnv *jvmti, JNIEnv *jni) { + ReferenceChainTracker::instance()->resolveLoadedClasses(jvmti, jni); + } + + static int lastResolvedClassCount() { + return ReferenceChainTracker::instance()->_last_resolved_class_count; + } + + // Durability re-verification test seams: direct pass-throughs + // to the private tie-break/rotation methods, plus FrontierTable::insert() + // itself (also private-by-convention here in the sense that production + // code only ever calls it via admitObject()) so tests can set up a + // frontier entry's exact starting root_kind/state/parent_tag without + // needing a live JVMTI mock for IterateOverReachableObjects/FollowReferences + // (neither is mocked in this file - see the file header's FollowReferences- + // only mock rationale). + static bool insertFrontierEntry(FrontierTable *frontier, jlong tag, + jlong parent_tag, u32 depth, u8 state, + u8 root_kind, u32 referrer_klass = 0, + jlong class_tag = 0, + jint referrer_field_index = -1, + u8 edge_kind = 0, + jlong referrer_class_tag = 0) { + return frontier->insert(tag, parent_tag, referrer_klass, depth, + state, root_kind, class_tag, + referrer_field_index, edge_kind, + referrer_class_tag); + } + + static bool maybeUpgradeRootAttachedRootKind(FrontierTable *frontier, + jlong tag, + u8 new_root_kind) { + return ReferenceChainTracker::instance() + ->maybeUpgradeRootAttachedRootKind(frontier, tag, new_root_kind); + } + + static std::vector collectStaleRootKindEntriesForRotation( + int max_count) { + return ReferenceChainTracker::instance() + ->collectStaleRootKindEntriesForRotation(max_count); + } + + static std::vector collectStaleExpandedEntriesForRotation( + int max_count) { + return ReferenceChainTracker::instance() + ->collectStaleExpandedEntriesForRotation(max_count); + } + + // Candidate-scoped reach (descendFromAnchor()/walkCandidateThreadLocals()/ + // walkStaticFieldAnchors()): direct drives for the same reason as + // expandFrontierForTest() above - a full runPass() drains a small graph + // to completion and its other phases add interference, so the walk + // phases are exercised on their own. + static void walkCandidateThreadLocalsForTest(jvmtiEnv *jvmti, JNIEnv *jni, + int budget, + int *edges_admitted) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + bool truncated = false; + bool cap_hit = false; + u64 safepoint_ticks = 0; + t->walkCandidateThreadLocals(jvmti, jni, budget, edges_admitted, + &truncated, &cap_hit, &safepoint_ticks); + } + + static void walkStaticFieldAnchorsForTest(jvmtiEnv *jvmti, JNIEnv *jni, + const std::vector &tags, + int budget, + int *edges_admitted) { + walkStaticAnchorFifoForTest(jvmti, jni, tags, budget, edges_admitted, + nullptr); + } + + static std::vector + collectStaticFieldAnchorsForRotationForTest(int max_count) { + return ReferenceChainTracker::instance() + ->collectStaticFieldAnchorsForRotation(max_count); + } + + static void addToStaticAnchorIndexForTest(jlong tag, jlong own_class_tag, + u8 root_kind) { + ReferenceChainTracker::instance() + ->addToStaticAnchorIndex(tag, own_class_tag, root_kind); + } + + // Prime the class-shape cache as if reconcileAnchorClassShapes() had + // classified `class_tag` (tests script shapes instead of driving the + // JNI interface walk, which needs real classes). + static void primeClassShapeForTest(jlong class_tag, bool container) { + ReferenceChainTracker::instance()->_class_shape_cache[class_tag] = + container + ? (u8)ReferenceChainTracker::AnchorClassShape::CONTAINER + : (u8)ReferenceChainTracker::AnchorClassShape::NON_CONTAINER; + } + + // B' at-risk static-anchor FIFO (see _static_anchor_fifo's declaration + // comment in referenceChains.h). pushStaticAnchorFifoForTest bypasses + // heapReferenceCallback's push predicate (kind == STATIC_FIELD onto a + // chain-attached entry) by calling the real pushAtRiskStaticAnchor() + // directly - the end-to-end push path is covered by + // AtRiskStaticHolderFeedsAnchorFifo below; the drive below exists so the + // drain/walk/requeue mechanics (and the round-16 per-class quota) can + // be exercised deterministically on seeded entries. + // Private nested names surfaced for TEST BODIES (friendship covers this + // class's own scope only, so test code cannot name them directly): + using AtRiskAnchor = ReferenceChainTracker::AtRiskAnchor; + static constexpr u32 kAtRiskPerKlassCap = + ReferenceChainTracker::STATIC_ANCHOR_ATRISK_PER_KLASS_CAP; + + static void pushStaticAnchorFifoForTest(jlong tag, u32 klass_id) { + ReferenceChainTracker::instance()->pushAtRiskStaticAnchor(tag, + klass_id); + } + + static int drainStaticAnchorFifoForTest( + int max_count, + std::vector &out) { + return ReferenceChainTracker::instance()->drainStaticAnchorFifo( + max_count, out); + } + + static void requeueStaticAnchorFifoFrontForTest( + const std::vector &entries) { + ReferenceChainTracker::instance()->requeueStaticAnchorFifoFront( + entries); + } + + static size_t staticAnchorFifoSizeForTest() { + return ReferenceChainTracker::instance()->_static_anchor_fifo.size(); + } + + static bool staticAnchorFifoContainsForTest(jlong tag) { + return ReferenceChainTracker::instance() + ->_static_anchor_fifo_set.contains(tag); + } + + static void walkStaticAnchorFifoForTest(jvmtiEnv *jvmti, JNIEnv *jni, + const std::vector &tags, + int budget, int *edges_admitted, + std::vector *unwalked) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + bool truncated = false; + bool cap_hit = false; + u64 safepoint_ticks = 0; + t->walkStaticFieldAnchors(jvmti, jni, tags, budget, edges_admitted, + &truncated, &cap_hit, &safepoint_ticks, + unwalked); + } + + // Direct candidate-slot seeding (the production path fills these via + // pollWatchedTargets()'s snapshot loop - see _candidate_qualifying_tids' + // own comment): the walk phase tests need exactly one (slot, klass, tid) + // combination without driving LivenessTracker's hysteresis machinery. + static void seedCandidateSlotForTest(int slot, u32 klass_id, + const jint *tids, int tid_count) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + t->_candidate_klass_ids[slot] = klass_id; + for (int i = 0; i < tid_count; i++) { + t->_candidate_qualifying_tids[slot][i] = tids[i]; + } + t->_candidate_qualifying_tid_count[slot] = tid_count; + if (slot + 1 > t->_candidate_count) { + t->_candidate_count = slot + 1; + } + } + + static int candidateQualifyingTidCountForTest(int slot) { + return ReferenceChainTracker::instance() + ->_candidate_qualifying_tid_count[slot]; + } + + static jlong getTagForTest(jvmtiEnv *jvmti, jobject obj) { + return ReferenceChainTracker::instance()->getTag(jvmti, obj); + } + + // Snapshot of _priority_expand's current contents, in queue order - used + // by tests to check for duplicate tags after both rotation collectors + // have run against it within the same simulated pass. + static std::vector priorityExpandContents() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return std::vector(t->_priority_expand.begin(), + t->_priority_expand.end()); + } + + // StaleExpandedRotationSkipsPreexistingQueueEntries below: simulates a + // tag left in _priority_expand by a prior pass's truncated expandFrontier() + // batch (expandFrontier()'s own "leave the batch at the front of the + // source queue for a later pass to retry" comment) without driving a full + // expandFrontier()/JVMTI round-trip to produce one. + static void pushPriorityExpand(jlong tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + t->_priority_expand.push_back(tag); + t->_priority_expand_set.insert(tag); + } + + // Simulates expandFrontier() having fully drained _priority_expand at the + // end of a pass (the common case: rotation's whole selection fit within + // that pass's rotation_budget slice) - see + // StaleExpandedRotationStarvesHighTagEntryBehindLowTagPopulation below, + // which needs this to model collectStaleExpandedEntriesForRotation() + // being called fresh on each of several simulated passes, the way + // runPassManualWalk() actually does it once per real pass. + static void clearPriorityExpand() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + t->_priority_expand.clear(); + t->_priority_expand_set.clear(); + } + + // Typed wrappers for PriorityExpandSetTest below - the set's + // probe/insert/rebuildFrom invariants are otherwise only exercised + // indirectly through admitObject()/rotation-collection paths. The + // wrapper methods live inside this friend accessor because the set's + // TYPE is a private nested class; the TEST bodies must not name it. + static bool pesInsert(jlong tag) { + return ReferenceChainTracker::instance()->_priority_expand_set.insert(tag); + } + static bool pesContains(jlong tag) { + return ReferenceChainTracker::instance()->_priority_expand_set.contains(tag); + } + static void pesClear() { + ReferenceChainTracker::instance()->_priority_expand_set.clear(); + } + static void pesRebuildFrom(const std::deque &queue) { + ReferenceChainTracker::instance()->_priority_expand_set.rebuildFrom(queue); + } + + static void setRootKindRotationCursor(jlong tag) { + ReferenceChainTracker::instance()->_root_kind_rotation_cursor = tag; + } + + static jlong rootKindRotationCursor() { + return ReferenceChainTracker::instance()->_root_kind_rotation_cursor; + } + + static int rootKindRotationBudget() { + return ReferenceChainTracker::ROOT_KIND_ROTATION_BUDGET; + } + + static int staleExpandedRotationBudget() { + return ReferenceChainTracker::STALE_EXPANDED_ROTATION_BUDGET; + } + + static size_t priorityExpandSize() { + return ReferenceChainTracker::instance()->_priority_expand.size(); + } + + // Snapshot of _pending_expand's current contents, in queue order - used + // by the rolling-resume smoke test to verify that a truncated + // expandFrontier() batch pops fully-processed entries (mark EXPANDED) + // and leaves only the partially-processed and unvisited entries at the + // front of the queue for the next pass to retry. + static std::vector pendingExpandContents() { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + return std::vector(t->_pending_expand.begin(), + t->_pending_expand.end()); + } + + static size_t pendingExpandSize() { + return ReferenceChainTracker::instance()->_pending_expand.size(); + } + + // Self-calibrating adaptive batch size (AIMD): read/write the per-call + // EMA and live batch size so tests can verify the AIMD dynamics. + static u64 gotwEmaCallNs() { + return ReferenceChainTracker::instance()->_gotw_ema_call_ns; + } + + static void setGotwEmaCallNs(u64 v) { + ReferenceChainTracker::instance()->_gotw_ema_call_ns = v; + } + + static size_t gotwBatchSize() { + return ReferenceChainTracker::instance()->_gotw_batch_size; + } + + static void setGotwBatchSize(size_t v) { + ReferenceChainTracker::instance()->_gotw_batch_size = v; + } + + // Read-only peeks at the batch-control constants (private statics - + // friendship applies inside this class's methods, not in test bodies). + static u64 gotwCpuBudgetNs() { + return ReferenceChainTracker::GOTW_CPU_BUDGET_NS; + } + + static size_t gotwInitialBatchSize() { + return (size_t)ReferenceChainTracker::GOTW_INITIAL_BATCH_SIZE; + } + + static size_t gotwMinBatch() { + return ReferenceChainTracker::GOTW_MIN_BATCH; + } + + static size_t gotwMaxBatch() { + return ReferenceChainTracker::GOTW_MAX_BATCH; + } + + static size_t gotwBacklogMinDepth() { + return ReferenceChainTracker::GOTW_BACKLOG_MIN_DEPTH; + } + + static u64 gotwBacklogWindowMult() { + return ReferenceChainTracker::GOTW_BACKLOG_WINDOW_MULT; + } + + // gotwWindowNs() is a pure function of (remaining window, lane depth) + // and the seeded EMA - directly unit-testable without a mock JVMTI call. + static u64 gotwWindowNs(u64 remaining_ns, size_t lane_depth) { + return ReferenceChainTracker::instance()->gotwWindowNs(remaining_ns, + lane_depth); + } + + static void setPassDeadlineNs(u64 v) { + ReferenceChainTracker::instance()->_pass_deadline_ns = v; + } + + static bool expandLanePreferPriority() { + return ReferenceChainTracker::instance()->_expand_lane_prefer_priority; + } + + // Leak-tag pool range base (private static) - same friend-access + // rationale as the AIMD constants above. + static jlong leakTagBase() { + return ReferenceChainTracker::LEAK_TAG_BASE; + } + + // Leak-accumulation rotation test seams (collectLeakAccumulationCandidatesForRotation()). + static void setWatchedLeakKlassIdsForTest(const std::vector &ids) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + int n = (int)std::min(ids.size(), + (size_t)ReferenceChainTracker::MAX_WATCHED_LEAK_KLASSES); + for (int i = 0; i < n; i++) { + t->_watched_leak_klass_ids[i] = ids[i]; + } + t->_watched_leak_klass_count = n; + } + + static void trackLeakAccumulation(FrontierTable *frontier, u32 referrer_klass, + jlong parent_tag, jlong tag) { + ReferenceChainTracker::instance()->trackLeakAccumulation( + frontier, referrer_klass, parent_tag, tag); + } + + static std::vector collectLeakAccumulationCandidatesForRotation( + int max_count) { + return ReferenceChainTracker::instance() + ->collectLeakAccumulationCandidatesForRotation(max_count); + } + + static int leakAccumulationRotationBudget() { + return ReferenceChainTracker::LEAK_ACCUMULATION_ROTATION_BUDGET; + } + + static u32 leakSignatureTotal(u32 leaf_klass_id, u32 parent_class_id) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + u64 key = t->leakSignatureKey(leaf_klass_id, parent_class_id); + auto it = t->_leak_signature_totals.find(key); + return it != t->_leak_signature_totals.end() ? it->second : 0; + } + + static u32 leakParentFanout(jlong parent_tag) { + ReferenceChainTracker *t = ReferenceChainTracker::instance(); + auto it = t->_leak_parent_fanout.find(parent_tag); + return it != t->_leak_parent_fanout.end() ? it->second.fanout : 0; + } + + static size_t leakSignatureCount() { + return ReferenceChainTracker::instance()->_leak_signature_totals.size(); + } + + static void seedLeakAccumulationForNewlyWatchedKlass(u32 klass_id) { + ReferenceChainTracker::instance() + ->seedLeakAccumulationForNewlyWatchedKlass(klass_id); + } +}; + +static jvmtiError JNICALL mock_SetEventNotificationMode(jvmtiEnv *, jvmtiEventMode, + jvmtiEvent, jthread, ...) { + return JVMTI_ERROR_NONE; +} + +class ReferenceChainsTest : public ::testing::Test { +protected: + jvmtiInterface_1_ tbl{}; + _jvmtiEnv mock_env{}; + jvmtiEnv *orig_jvmti = nullptr; + + void SetUp() override { + orig_jvmti = VMTestAccessor::getJvmti(); + tbl = jvmtiInterface_1_{}; + tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + mock_env.functions = &tbl; + VMTestAccessor::setJvmti(&mock_env); + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + } +}; + +TEST_F(ReferenceChainsTest, DefaultDisabled) { + Arguments args; + EXPECT_FALSE(args._reference_chains); +} + +TEST_F(ReferenceChainsTest, FlagParsesEnabled) { + Arguments args; + Error error = args.parse("referencechains=true"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); +} + +TEST_F(ReferenceChainsTest, FlagParsesDisabled) { + Arguments args; + Error error = args.parse("referencechains=false"); + EXPECT_FALSE(error); + EXPECT_FALSE(args._reference_chains); +} + +TEST_F(ReferenceChainsTest, FlagParsesSubOptions) { + Arguments args; + Error error = args.parse("referencechains=true:hops=64:budget=2000:ttl=5000:framecap=128"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); + EXPECT_EQ(64, args._reference_chains_hop_cap); + EXPECT_EQ(2000, args._reference_chains_budget); + EXPECT_EQ(5000, args._reference_chains_ttl_ms); + EXPECT_EQ(128, args._reference_chains_frontier_cap); +} + +// Negative/out-of-range sub-options must be floored/clamped at the parse +// boundary (Arguments::parse(), arguments.cpp) rather than stored verbatim - +// see that call site's own comment for why an unclamped negative hops in +// particular is dangerous: `depth >= (u32)ctx->hop_cap` (referenceChains.cpp) +// casts a negative int to u32, wrapping to ~4e9 and silently disabling the +// hop cap entirely. +TEST_F(ReferenceChainsTest, FlagClampsNegativeSubOptions) { + Arguments args; + Error error = args.parse( + "referencechains=true:hops=-1:budget=-5:ttl=-1:framecap=-3:" + "pausetarget=-1:painbudget=-10"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); + // Floored to a sane minimum (1), not left negative - a negative value + // cast to u32 downstream would otherwise wrap to a huge positive number. + EXPECT_GT(args._reference_chains_hop_cap, 0); + EXPECT_GT(args._reference_chains_budget, 0); + EXPECT_GT(args._reference_chains_frontier_cap, 0); + // ttl/pausetarget are floored at 0 (their own downstream gates already + // treat 0 as "disabled", so 0 - not 1 - is the correct floor). + EXPECT_GE(args._reference_chains_ttl_ms, 0); + EXPECT_GE(args._reference_chains_pause_target_ms, 0); + // painbudget is a percentage - clamped into [0, 100]. + EXPECT_GE(args._reference_chains_pain_budget_percent, 0); + EXPECT_LE(args._reference_chains_pain_budget_percent, 100); +} + +// A too-large painbudget must be clamped down to 100, not stored verbatim - +// the sibling of FlagClampsNegativeSubOptions above, for the upper bound +// rather than the lower one. +TEST_F(ReferenceChainsTest, FlagClampsOversizedPainBudgetPercent) { + Arguments args; + Error error = args.parse("referencechains=true:painbudget=250"); + EXPECT_FALSE(error); + EXPECT_EQ(100, args._reference_chains_pain_budget_percent); +} + +TEST_F(ReferenceChainsTest, FlagWithOtherArgsDoesNotClobberOuterParse) { + Arguments args; + Error error = args.parse("event=cpu,referencechains=true:hops=32,interval=1000000"); + EXPECT_FALSE(error); + EXPECT_TRUE(args._reference_chains); + EXPECT_EQ(32, args._reference_chains_hop_cap); + EXPECT_STREQ("cpu", args._event); + EXPECT_EQ(1000000, args._interval); +} + +TEST_F(ReferenceChainsTest, StartStopDisabledDoesNotCrash) { + Arguments args; + Error error = args.parse("referencechains=false"); + ASSERT_FALSE(error); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + Error startError = tracker->start(args); + EXPECT_FALSE(startError); + EXPECT_FALSE(tracker->enabled()); + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, StartStopEnabledDoesNotCrash) { + Arguments args; + Error error = args.parse("referencechains=true:hops=10:budget=100"); + ASSERT_FALSE(error); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + Error startError = tracker->start(args); + EXPECT_FALSE(startError); + EXPECT_TRUE(tracker->enabled()); + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// GC signal (GarbageCollectionStart/Finish -> epoch counters). +// +// The callback trampolines (ReferenceChainTracker::GarbageCollectionStart/ +// Finish) ignore the jvmtiEnv* argument entirely - onGCStart()/onGCFinish() +// only bump an atomic counter, per the JVMTI spec restriction documented in +// referenceChains.h - so passing nullptr here exercises the real production +// code path. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsTest, GCCallbacksIncrementEpochWhenEnabled) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + u64 startBefore = tracker->gcStartEpoch(); + u64 finishBefore = tracker->gcFinishEpoch(); + + ReferenceChainTracker::GarbageCollectionStart(nullptr); + ReferenceChainTracker::GarbageCollectionFinish(nullptr); + + EXPECT_EQ(startBefore + 1, tracker->gcStartEpoch()); + EXPECT_EQ(finishBefore + 1, tracker->gcFinishEpoch()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, GCCallbacksAreNoOpWhenDisabled) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=false")); + + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + ASSERT_FALSE(tracker->enabled()); + + u64 startBefore = tracker->gcStartEpoch(); + u64 finishBefore = tracker->gcFinishEpoch(); + + ReferenceChainTracker::GarbageCollectionStart(nullptr); + ReferenceChainTracker::GarbageCollectionFinish(nullptr); + + EXPECT_EQ(startBefore, tracker->gcStartEpoch()); + EXPECT_EQ(finishBefore, tracker->gcFinishEpoch()); +} + +// --------------------------------------------------------------------------- +// Tag round-trip (SetTag/GetTag/clear). +// +// The natural smoke test ("allocate an object, tag it, +// force a GC, confirm the tag is still readable via GetObjectsWithTags") +// assumes a live embedded JVM. This gtest binary has no live JVM attached +// (see jvmSupport_ut.cpp's fixture comment for the same constraint on a +// different subsystem), so - following this repo's established pattern for +// testing JVMTI call sites without a real JVM (objectSampler_ut.cpp's mock +// jvmtiInterface_1_ table) - these tests exercise tagObject()/getTag()/ +// clearTag() against a mock jvmtiEnv backed by an in-memory tag map, rather +// than a real GC. This proves the SetTag/GetTag/SetTag(obj,0) call sequence +// and unique-tag allocation are correct; it does not prove GC-move- +// transparency, which requires a real collector and is out of reach of this +// native-only gtest binary. +// --------------------------------------------------------------------------- + +class ReferenceChainsTagTest : public ::testing::Test { +protected: + jvmtiInterface_1_ tbl{}; + _jvmtiEnv mock_env{}; + std::unordered_map tags; + + static ReferenceChainsTagTest *active_fixture; + + void SetUp() override { + active_fixture = this; + tbl = jvmtiInterface_1_{}; + tbl.SetTag = &mock_SetTag; + tbl.GetTag = &mock_GetTag; + mock_env.functions = &tbl; + } + + void TearDown() override { + active_fixture = nullptr; + } + + static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { + if (tag == 0) { + active_fixture->tags.erase(object); + } else { + active_fixture->tags[object] = tag; + } + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { + auto it = active_fixture->tags.find(object); + *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; + return JVMTI_ERROR_NONE; + } +}; + +ReferenceChainsTagTest *ReferenceChainsTagTest::active_fixture = nullptr; + +TEST_F(ReferenceChainsTagTest, TagRoundTripsThenClears) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + + jlong tag = tracker->tagObject(&mock_env, obj); + EXPECT_NE(0, tag); + EXPECT_EQ(tag, tracker->getTag(&mock_env, obj)); + + tracker->clearTag(&mock_env, obj); + EXPECT_EQ(0, tracker->getTag(&mock_env, obj)); +} + +TEST_F(ReferenceChainsTagTest, TagsAreUniqueAndNeverZero) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + int a = 0, b = 0; + jlong tagA = tracker->tagObject(&mock_env, reinterpret_cast(&a)); + jlong tagB = tracker->tagObject(&mock_env, reinterpret_cast(&b)); + + EXPECT_NE(0, tagA); + EXPECT_NE(0, tagB); + EXPECT_NE(tagA, tagB); +} + +TEST_F(ReferenceChainsTagTest, UntaggedObjectReadsBackZero) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + int untagged = 0; + EXPECT_EQ(0, tracker->getTag(&mock_env, reinterpret_cast(&untagged))); +} + +// --------------------------------------------------------------------------- +// FrontierTable (tag-indexed frontier metadata table). +// +// No live JVM/JVMTI involvement here - FrontierTable is pure native slot +// storage indexed by an already-issued tag value, so these tests exercise it +// directly rather than through ReferenceChainTracker's tag helpers. +// --------------------------------------------------------------------------- + +TEST(FrontierTableTest, InsertThenLookupRoundTrips) { + FrontierTable table(64); + + ASSERT_TRUE(table.insert(1, /*parent_tag=*/0, /*referrer_klass=*/7, + /*depth=*/0, FrontierEntryState::FRONTIER)); + + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(1, &entry)); + EXPECT_EQ(0, entry.parent_tag); + EXPECT_EQ(7u, entry.referrer_klass); + EXPECT_EQ(0u, entry.depth); + EXPECT_EQ(FrontierEntryState::FRONTIER, entry.state); +} + +TEST(FrontierTableTest, LookupOfNeverInsertedTagFails) { + FrontierTable table(64); + FrontierEntry entry{}; + EXPECT_FALSE(table.lookup(1, &entry)); + EXPECT_FALSE(table.lookup(5, &entry)); +} + +TEST(FrontierTableTest, NonPositiveTagIsRejected) { + FrontierTable table(64); + FrontierEntry entry{}; + EXPECT_FALSE(table.insert(0, 0, 0, 0)); + EXPECT_FALSE(table.insert(-1, 0, 0, 0)); + EXPECT_FALSE(table.lookup(0, &entry)); + EXPECT_FALSE(table.lookup(-1, &entry)); +} + +TEST(FrontierTableTest, LookupLockedRejectsNonPositiveTag) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); + FrontierEntry entry{}; + // tag=0 must be rejected the same way lookup() rejects it - callers + // index the table with tag-1, so a `tag <= 0` check (not just `tag < 0`) + // is required to keep that subtraction from wrapping into a valid slot. + EXPECT_FALSE(table.lookupLocked(0, &entry)); + EXPECT_FALSE(table.lookupLocked(-1, &entry)); +} + +TEST(FrontierTableTest, LookupLockedRejectsTagPastCurrentSize) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); + FrontierEntry entry{}; + // Only tag=1 has ever been inserted (table_size == 1); tag=2 maps to + // idx=1, exactly at the current size boundary, and must be rejected + // rather than read out of bounds. + EXPECT_FALSE(table.lookupLocked(2, &entry)); +} + +TEST(FrontierTableTest, ParentTagChainReconstructsAcrossHops) { + // Mirrors how the heap-walk engine walks parent_tag links back to a root: insert + // a small chain root(tag=1) <- mid(tag=2) <- leaf(tag=3) and confirm the + // links resolve in order. + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 100, 0, FrontierEntryState::EDGE)); + ASSERT_TRUE(table.insert(2, 1, 200, 1, FrontierEntryState::EDGE)); + ASSERT_TRUE(table.insert(3, 2, 300, 2, FrontierEntryState::EDGE)); + + FrontierEntry entry{}; + jlong tag = 3; + std::vector chain; + while (tag != 0) { + ASSERT_TRUE(table.lookup(tag, &entry)); + chain.push_back(entry.referrer_klass); + tag = entry.parent_tag; + } + + ASSERT_EQ(3u, chain.size()); + EXPECT_EQ(300u, chain[0]); + EXPECT_EQ(200u, chain[1]); + EXPECT_EQ(100u, chain[2]); +} + +TEST(FrontierTableTest, ClearMarksAbandonedWithoutRemovingEntry) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 7, 0, FrontierEntryState::FRONTIER)); + + table.clear(1); + + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(1, &entry)); + EXPECT_EQ(FrontierEntryState::ABANDONED, entry.state); +} + +TEST(FrontierTableTest, ClearOfNeverInsertedTagIsNoOp) { + FrontierTable table(64); + table.clear(1); // must not crash + FrontierEntry entry{}; + EXPECT_FALSE(table.lookup(1, &entry)); +} + +TEST(FrontierTableTest, GrowsPastInitialCapacityUpToMaxCap) { + // Force at least one resize by inserting beyond the small max_cap. + const int max_cap = 10; + FrontierTable table(max_cap); + ASSERT_LE(table.capacity(), max_cap); + + for (jlong tag = 1; tag <= max_cap; tag++) { + ASSERT_TRUE(table.insert(tag, tag - 1, (u32)tag, (u32)(tag - 1))) + << "insert failed for tag " << tag; + } + EXPECT_EQ(max_cap, table.capacity()); + + for (jlong tag = 1; tag <= max_cap; tag++) { + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(tag, &entry)); + EXPECT_EQ((u32)tag, entry.referrer_klass); + } +} + +TEST(FrontierTableTest, CapacityExhaustedReportsFailureInsteadOfCrashing) { + const int max_cap = 4; + FrontierTable table(max_cap); + + for (jlong tag = 1; tag <= max_cap; tag++) { + ASSERT_TRUE(table.insert(tag, 0, 0, 0)); + } + // One past max_cap must be rejected, not silently dropped-but-crashing. + EXPECT_FALSE(table.insert(max_cap + 1, 0, 0, 0)); + EXPECT_EQ(max_cap, table.capacity()); + + // Existing entries remain intact after the failed insert. + FrontierEntry entry{}; + EXPECT_TRUE(table.lookup(1, &entry)); +} + +TEST(FrontierTableTest, ZeroMaxCapRejectsEveryInsert) { + FrontierTable table(0); + EXPECT_EQ(0, table.capacity()); + EXPECT_FALSE(table.insert(1, 0, 0, 0)); +} + +TEST(FrontierTableTest, ConcurrentInsertWhileGrowingDoesNotCrash) { + // Small max_cap relative to thread/tag count forces repeated resizes + // while other threads are concurrently inserting distinct tags. + const int max_cap = 4096; + const int thread_count = 8; + const int tags_per_thread = 256; + FrontierTable table(max_cap); + + std::vector threads; + for (int t = 0; t < thread_count; t++) { + threads.emplace_back([&table, t, tags_per_thread]() { + for (int i = 0; i < tags_per_thread; i++) { + jlong tag = (jlong)t * tags_per_thread + i + 1; + table.insert(tag, 0, (u32)tag, 0); + } + }); + } + for (auto &th : threads) { + th.join(); + } + + int found = 0; + for (jlong tag = 1; tag <= (jlong)thread_count * tags_per_thread; tag++) { + FrontierEntry entry{}; + if (table.lookup(tag, &entry)) { + EXPECT_EQ((u32)tag, entry.referrer_klass); + found++; + } + } + // Every tag fits well within max_cap, so all inserts must have + // succeeded and be independently readable. + EXPECT_EQ(thread_count * tags_per_thread, found); +} + +TEST(FrontierTableTest, ReconstructChainWalksParentTagsAndMarksEdge) { + FrontierTable table(64); + ASSERT_TRUE(table.insert(1, 0, 100, 0, FrontierEntryState::FRONTIER)); + ASSERT_TRUE(table.insert(2, 1, 200, 1, FrontierEntryState::FRONTIER)); + ASSERT_TRUE(table.insert(3, 2, 300, 2, FrontierEntryState::FRONTIER)); + + std::vector chain; + ASSERT_TRUE(table.reconstructChain(3, &chain)); + ASSERT_EQ(3u, chain.size()); + EXPECT_EQ(300u, chain[0]); + EXPECT_EQ(200u, chain[1]); + EXPECT_EQ(100u, chain[2]); + + // Every hop walked must be marked EDGE - this table's degenerate + // EdgeStore (design doc: "on a path toward a target sample"). + for (jlong tag = 1; tag <= 3; tag++) { + FrontierEntry entry{}; + ASSERT_TRUE(table.lookup(tag, &entry)); + EXPECT_EQ(FrontierEntryState::EDGE, entry.state); + } +} + +TEST(FrontierTableTest, ReconstructChainOfNeverInsertedTagFails) { + FrontierTable table(64); + std::vector chain; + EXPECT_FALSE(table.reconstructChain(1, &chain)); +} + +// --------------------------------------------------------------------------- +// Heap-walk engine (ReferenceChainTracker::runPass()/ +// heapReferenceCallback()/resolveLoadedClasses()). +// +// The design doc's suggestion is to test this against "a small live-object +// graph in a test JVM (via JNI from the test)". This native-only gtest +// binary has no live JVM at all (see this file's GC-signal/tag-round-trip +// comment above, and jvmSupport_ut.cpp's fixture comment, for the same +// pre-existing constraint) - no gtest binary in this codebase embeds a +// JNI_CreateJavaVM-created JVM. So, exactly like ReferenceChainsTagTest above +// mocks SetTag/GetTag with an in-memory map, these tests mock the JVMTI/JNI +// call boundary (FollowReferences/GetLoadedClasses/GetClassSignature/ +// DeleteLocalRef) to play back a scripted synthetic object graph, and run +// the *real* production heapReferenceCallback()/resolveLoadedClasses()/ +// reconstructChain() code against it - only the JVMTI/JNI calls are faked, +// not the logic under test. A live-JVM end-to-end test belongs to the +// Java-side integration suite, not this native gtest binary. +// --------------------------------------------------------------------------- + +namespace { + +struct ScriptedEdge { + jvmtiHeapReferenceKind kind; + int referrer_idx; // -1 = heap root (no referrer) + int referee_idx; // index into ReferenceChainsBfsTest::node_tags + int class_idx; // index into ReferenceChainsBfsTest::classes, or -1 +}; + +struct ScriptedClass { + void *klass; + const char *signature; // JVMTI class signature, e.g. "Lcom/example/Foo;" +}; + +// Retention-edge label decode fixtures (ReferenceChainsBfsTest's +// field_decode_hierarchy + the hierarchy-introspection mock slots): a fake +// class hierarchy the slots read, mirroring just enough JVMTI class shape +// for the spec-ordinal decoder (own-declared fields in GetClassFields +// order, direct superclass, directly implemented/extended interfaces). +struct FakeField { + void *id; // fake jfieldID + const char *name; +}; +struct FakeClass { + bool is_interface; + void *super; // fake jclass, or nullptr + std::vector interfaces; // directly implemented/extended + std::vector fields; // own-declared, GetClassFields order +}; + +} // namespace + +class ReferenceChainsBfsTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + JNINativeInterface_ jni_tbl{}; + JNIEnv_ mock_jni{}; + + std::unordered_map tags; + std::vector classes; + std::vector script; + std::vector node_tags; + + // node_tags[idx] mirrors "the object's *current* live JVMTI + // tag" (0 once releaseSearchTags() clears it, exactly like a real + // GetTag() would report after SetTag(obj, 0)). tags_ever_assigned[idx] + // instead remembers the tag heapReferenceCallback() ever wrote through + // tag_ptr for this node, and is never reset - a production consumer + // would capture a target sample's tag the same way (at assignment time, + // e.g. via its own sample-tracking), not by re-reading GetTag() after + // the search has already released it. Tests use this to fetch a tag for + // reconstructChain() without depending on whether the search released + // it before or after the test could observe node_tags[idx]. + std::vector tags_ever_assigned; + + // Tags that GetObjectsWithTags() below reports as unresolvable, + // simulating the referenced object having died (GC'd) between passes - + // see the resolve-or-drop tests. + std::unordered_set dead_tags; + + // When true, mock_GetObjectsWithTags() below fails outright (as if the + // real JVMTI call had hit e.g. JVMTI_ERROR_OUT_OF_MEMORY), for + // ReleaseSearchTagsFailureTest - simulates releaseSearchTags()'s own + // GetObjectsWithTags() call failing rather than an individual tag + // failing to resolve (dead_tags above). + bool fail_get_objects_with_tags = false; + + // When non-zero, mock_GetObjectsWithTags() below busy-waits this many + // nanoseconds. The mock call is otherwise ~free, so a pass deadline set + // to a fraction of this value bounds an expandFrontier() invocation to + // exactly ONE batch - the production regime (one real ~25-30ms call of + // a 50ms window), needed by the lane-alternation test. + u64 gotw_delay_ns = 0; + + // Synthetic frontier-holder arrays for expandFrontier()'s array-holder + // walk: mock_NewObjectArray() hands back an opaque handle, + // mock_SetObjectArrayElement() records its elements here, and + // mock_FollowReferences() treats every recorded element as an expansion + // seed (one hop, gated by the production callback's batch_tags) when the + // holder is passed as initial_object. + std::unordered_map> holders; + uintptr_t next_holder = 0xF00D0000; + + // FindClass(name) -> registered fake class (see mock_FindClass' own + // comment): names descendFromAnchor()'s resolutions look up + // ("java/lang/ClassLoader", "java/lang/ThreadGroup", + // "java/security/ProtectionDomain", + // "java/lang/ThreadLocal$ThreadLocalMap", "java/lang/Thread"). + std::unordered_map find_classes; + // Fake class returned by mock_GetObjectClass() for unregistered objects + // (walkCandidateThreadLocals()'s fresh-anchor admission path). + void *thread_class = nullptr; + + // DeleteGlobalRef call count (see mock_DeleteGlobalRef). + int global_refs_deleted_ = 0; + + jvmtiEnv *orig_jvmti = nullptr; + + static ReferenceChainsBfsTest *active_fixture; + + void SetUp() override { + active_fixture = this; + // See ReferenceChainsTestAccessor's own comment - without this, a + // prior test in this suite that drove the search to + // SearchState::COMPLETED/ABANDONED would make every runPass() call + // below a permanent no-op. + ReferenceChainsTestAccessor::reset(); + jvmti_tbl = jvmtiInterface_1_{}; + // start() calls VM::jvmti()->SetEventNotificationMode() - + // stub it and swap VM::_jvmti (VMTestAccessor, declared above) the + // same way ReferenceChainsTest's fixture does, so start() does not + // dereference the real (null, no live JVM) jvmtiEnv. + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.SetTag = &mock_SetTag; + jvmti_tbl.GetTag = &mock_GetTag; + jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; + jvmti_tbl.GetClassLoader = &mock_GetClassLoader; + jvmti_tbl.GetClassSignature = &mock_GetClassSignature; + jvmti_tbl.Deallocate = &mock_Deallocate; + jvmti_tbl.FollowReferences = &mock_FollowReferences; + jvmti_tbl.IterateOverReachableObjects = &mock_IterateOverReachableObjects; + jvmti_tbl.GetObjectsWithTags = &mock_GetObjectsWithTags; + // Retention-edge label decode path (hopLabelClassFor()). + jvmti_tbl.IsInterface = &mock_IsInterface; + jvmti_tbl.GetImplementedInterfaces = &mock_GetImplementedInterfaces; + jvmti_tbl.GetClassFields = &mock_GetClassFields; + jvmti_tbl.GetFieldName = &mock_GetFieldName; + mock_jvmti.functions = &jvmti_tbl; + orig_jvmti = VMTestAccessor::getJvmti(); + VMTestAccessor::setJvmti(&mock_jvmti); + + jni_tbl = JNINativeInterface_{}; + jni_tbl.DeleteLocalRef = &mock_DeleteLocalRef; + jni_tbl.FindClass = &mock_FindClass; + jni_tbl.GetObjectClass = &mock_GetObjectClass; + jni_tbl.GetSuperclass = &mock_JniGetSuperclass; + jni_tbl.NewGlobalRef = &mock_NewGlobalRef; + jni_tbl.DeleteGlobalRef = &mock_DeleteGlobalRef; + jni_tbl.EnsureLocalCapacity = &mock_EnsureLocalCapacity; + jni_tbl.NewObjectArray = &mock_NewObjectArray; + jni_tbl.SetObjectArrayElement = &mock_SetObjectArrayElement; + jni_tbl.ExceptionCheck = &mock_ExceptionCheck; + jni_tbl.ExceptionClear = &mock_ExceptionClear; + mock_jni.functions = &jni_tbl; + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + active_fixture = nullptr; + } + + // Registers a fake class (matched by identity, not by any real JNI + // semantics) that resolveLoadedClasses() will discover via the mocked + // GetLoadedClasses(). Returns its index into `classes`. + int addClass(void *klass, const char *signature) { + classes.push_back({klass, signature}); + return (int)classes.size() - 1; + } + + // addClass() + a mock_FindClass(name) registry entry in one step, for + // the classes descendFromAnchor()'s resolution helpers look up by + // name (see find_classes' own comment). `name` uses FindClass's + // binary-name form ("java/lang/ClassLoader"), `signature` the + // class-signature form resolveLoadedClasses() interns + // ("Ljava/lang/ClassLoader;") - the production code passes each to + // exactly one of the two APIs. + int registerClassForFindClass(void *klass, const char *name, + const char *signature) { + int idx = addClass(klass, signature); + find_classes[name] = klass; + return idx; + } + + // Adds an as-yet-untagged frontier node, returning its index into + // node_tags for use as a ScriptedEdge referrer_idx/referee_idx. + int addNode() { + node_tags.push_back(0); + tags_ever_assigned.push_back(0); + return (int)node_tags.size() - 1; + } + + // Reverse lookup from a node's synthetic identity + // (&node_tags[idx], see mock_FollowReferences' initial_object handling + // below) back to its index. Returns -1 for anything else (e.g. a + // ScriptedClass's `klass` pointer, which never aliases node_tags' + // backing storage). Requires every addNode() call to happen before any + // runPass() call in a test, so node_tags never reallocates out from + // under a previously-taken address - true of every test in this file. + int indexOfNode(jobject obj) const { + for (size_t i = 0; i < node_tags.size(); i++) { + if (obj == (jobject)&node_tags[i]) { + return (int)i; + } + } + return -1; + } + + static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { + // releaseSearchTags() calls SetTag(obj, 0) on the resolved + // objects GetObjectsWithTags() (below) hands back for a frontier + // node - route that through node_tags[idx] directly (the same + // storage GetObjectsWithTags's resolution and the production + // callback's tag_ptr writes both key off of), so the release is + // actually observable, not just recorded in a side map nothing else + // reads. + int idx = active_fixture->indexOfNode(object); + if (idx >= 0) { + active_fixture->node_tags[idx] = tag; + return JVMTI_ERROR_NONE; + } + if (tag == 0) { + active_fixture->tags.erase(object); + } else { + active_fixture->tags[object] = tag; + } + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { + int idx = active_fixture->indexOfNode(object); + if (idx >= 0) { + *tag_ptr = active_fixture->node_tags[idx]; + return JVMTI_ERROR_NONE; + } + auto it = active_fixture->tags.find(object); + *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count_ptr, + jclass **classes_ptr) { + auto &classes = active_fixture->classes; *count_ptr = (jint)classes.size(); + *classes_ptr = classes.empty() + ? nullptr + : (jclass *)malloc(sizeof(jclass) * classes.size()); + for (size_t i = 0; i < classes.size(); i++) { + (*classes_ptr)[i] = (jclass)classes[i].klass; + } + return JVMTI_ERROR_NONE; + } + + // admitStaticFieldRoots()'s app-classes-first partition (referenceChains.cpp) + // calls this for every loaded class. Every fixture class is "bootstrap" + // (null classloader) so the partition is a no-op and this suite's + // scripted class order/indices stay exactly as each test set them up. + static jvmtiError JNICALL mock_GetClassLoader(jvmtiEnv *, jclass, + jobject *classloader_ptr) { + *classloader_ptr = nullptr; + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_GetClassSignature(jvmtiEnv *, jclass klass, + char **signature_ptr, + char **generic_ptr) { + for (auto &c : active_fixture->classes) { + if (c.klass == (void *)klass) { + *signature_ptr = strdup(c.signature); + if (generic_ptr != nullptr) { + *generic_ptr = nullptr; + } + return JVMTI_ERROR_NONE; + } + } + return JVMTI_ERROR_INVALID_CLASS; + } + + static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { + free(mem); + return JVMTI_ERROR_NONE; + } + + static void JNICALL mock_DeleteLocalRef(JNIEnv *, jobject) { + // no-op: this fixture's fake jobject/jclass values are not real JNI + // local refs. + } + + // Retention-edge label decode fixtures (fillHopEdgeLabels()/ + // hopLabelClassFor()): a fake class hierarchy the hierarchy-introspection + // slots below read, plus a tag -> fake jclass map (the decoder resolves + // the referrer class from its raw tag via GetObjectsWithTags - the test + // populates field_decode_classes from resolveLoadedClasses()-minted + // tags, and mock_GetObjectsWithTags consults it first). + std::unordered_map field_decode_classes; + std::unordered_map field_decode_hierarchy; + + // The hierarchy-introspection slots the decoder needs (IsInterface/ + // GetImplementedInterfaces/GetClassFields/GetFieldName on the JVMTI + // table, GetSuperclass on the JNI table - modern JVMTI dropped its own + // GetSuperclass). All lookups go through field_decode_hierarchy; an + // unregistered class returns a failure code so the decoder marks the + // class undecodable and degrades to kind labels - the exact production + // fail-safe shape. + static jvmtiError JNICALL mock_IsInterface(jvmtiEnv *, jclass cls, + jboolean *is_interface_ptr) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + if (it == active_fixture->field_decode_hierarchy.end()) { + return JVMTI_ERROR_INVALID_CLASS; + } + *is_interface_ptr = it->second.is_interface ? JNI_TRUE : JNI_FALSE; + return JVMTI_ERROR_NONE; + } + static jclass JNICALL mock_JniGetSuperclass(JNIEnv *, jclass cls) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + return it == active_fixture->field_decode_hierarchy.end() + ? nullptr + : (jclass)it->second.super; + } + static jvmtiError JNICALL mock_GetImplementedInterfaces( + jvmtiEnv *, jclass cls, jint *count_ptr, jclass **ifaces_ptr) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + if (it == active_fixture->field_decode_hierarchy.end()) { + return JVMTI_ERROR_INVALID_CLASS; + } + const std::vector &ifaces = it->second.interfaces; + *count_ptr = (jint)ifaces.size(); + *ifaces_ptr = ifaces.empty() + ? nullptr + : (jclass *)malloc(sizeof(jclass) * ifaces.size()); + for (size_t i = 0; i < ifaces.size(); i++) { + (*ifaces_ptr)[i] = (jclass)ifaces[i]; + } + return JVMTI_ERROR_NONE; + } + static jvmtiError JNICALL mock_GetClassFields( + jvmtiEnv *, jclass cls, jint *count_ptr, jfieldID **fields_ptr) { + auto it = active_fixture->field_decode_hierarchy.find(cls); + if (it == active_fixture->field_decode_hierarchy.end()) { + return JVMTI_ERROR_INVALID_CLASS; + } + const std::vector &fields = it->second.fields; + *count_ptr = (jint)fields.size(); + *fields_ptr = fields.empty() + ? nullptr + : (jfieldID *)malloc(sizeof(jfieldID) * fields.size()); + for (size_t i = 0; i < fields.size(); i++) { + (*fields_ptr)[i] = (jfieldID)fields[i].id; + } + return JVMTI_ERROR_NONE; + } + static jvmtiError JNICALL mock_GetFieldName( + jvmtiEnv *, jclass, jfieldID field, char **name_ptr, + char ** /*signature_ptr*/, char ** /*generic_ptr*/) { + for (const auto &kv : active_fixture->field_decode_hierarchy) { + for (const FakeField &f : kv.second.fields) { + if (f.id == (void *)field) { + size_t len = strlen(f.name) + 1; + char *name = (char *)malloc(len); + memcpy(name, f.name, len); + *name_ptr = name; + return JVMTI_ERROR_NONE; + } + } + } + return JVMTI_ERROR_INVALID_FIELDID; + } + + // expandFrontier()/admitStaticFieldRoots() resolve java/lang/Object once + // as the holder array's element type - a non-null fake jclass is all it + // needs (the type is never introspected, only passed to NewObjectArray()). + // descendFromAnchor()'s class resolution (resolveNoDescendClassTags()/ + // resolveThreadLocalMapClassTag()) additionally needs NAME lookup - the + // find_classes registry maps a FindClass name to a class registered via + // addClass() so those resolutions see the same tagged fake class the + // scripted graph uses. + static jclass JNICALL mock_FindClass(JNIEnv *, const char *name) { + auto it = active_fixture->find_classes.find(name); + if (it != active_fixture->find_classes.end()) { + return (jclass)it->second; + } + return (jclass)0xC1A55; + } + + // walkCandidateThreadLocals()'s fresh-anchor admission calls + // GetObjectClass(thread) - unregistered classes return the fixture's + // fake Thread class (set_thread_class) so the anchor entry's class tag + // resolves through the same mocked GetTag/tagging path. + static jclass JNICALL mock_GetObjectClass(JNIEnv *, jobject) { + return (jclass)active_fixture->thread_class; + } + + + // The production code wraps that fake jclass in a global ref (a real + // local ref would dangle across JNI-entered test seams - see + // _cached_object_class's own comment). This fixture's "refs" are raw + // fake pointers with no JNI lifetime, so identity is the correct mock. + static jobject JNICALL mock_NewGlobalRef(JNIEnv *, jobject obj) { + return obj; + } + + // Counts DeleteGlobalRef calls - the deferred thread-ref teardown test + // (ThreadRefUnregisterDefersGlobalRefDeletion) asserts on the count. + static void JNICALL mock_DeleteGlobalRef(JNIEnv *, jobject) { + active_fixture->global_refs_deleted_++; + } + + static jint JNICALL mock_EnsureLocalCapacity(JNIEnv *, jint) { + return JNI_OK; + } + + // expandFrontier() calls jniExceptionCheck() after every upcall that can + // legally throw (NewObjectArray/SetObjectArrayElement/EnsureLocalCapacity + // failures) - this fixture's mocks never throw, so there is never a + // pending exception to report or clear. + static jboolean JNICALL mock_ExceptionCheck(JNIEnv *) { + return JNI_FALSE; + } + + static void JNICALL mock_ExceptionClear(JNIEnv *) { + // no-op: mock_ExceptionCheck() never reports a pending exception. + } + + // Hands back a fresh opaque holder handle and registers it in `holders` so + // mock_SetObjectArrayElement()/mock_FollowReferences() can find its + // elements. `length`/`elementClass`/`initialElement` are unused - the + // fixture never reads the array back, only its recorded element list. + static jobjectArray JNICALL mock_NewObjectArray(JNIEnv *, jsize, jclass, + jobject) { + jobject handle = (jobject)(active_fixture->next_holder++); + active_fixture->holders[handle] = {}; + return (jobjectArray)handle; + } + + static void JNICALL mock_SetObjectArrayElement(JNIEnv *, jobjectArray array, + jsize idx, jobject value) { active_fixture->holders[(jobject)array].push_back(value); + } + + // runPassManualWalk()'s root enumeration (the default, non-fallback path): + // reports each scripted root edge's referee to heapRootCallback() exactly + // as a real IterateOverReachableObjects() reports a root-held object - + // tag_ptr only, no oop, no transitive children (see runPassManualWalk()'s + // own comment). Expansion past the roots is then driven by expandFrontier() + // through the same mock_FollowReferences() array-holder path the resumed + // fallback passes use. stack_ref/object_ref callbacks are unused here - a + // JNI-global root (durable) is all these tests need to model. + static jvmtiError JNICALL mock_IterateOverReachableObjects( + jvmtiEnv *, jvmtiHeapRootCallback heap_root_cb, + jvmtiStackReferenceCallback, jvmtiObjectReferenceCallback, + const void *user_data) { + for (auto &e : active_fixture->script) { + if (e.referrer_idx != -1) { + continue; + } + jlong class_tag = 0; + if (e.class_idx >= 0) { + class_tag = active_fixture->tags[active_fixture->classes[e.class_idx].klass]; + } + jlong *tag_ptr = &active_fixture->node_tags[e.referee_idx]; + jvmtiIterationControl ctl = heap_root_cb( + JVMTI_HEAP_ROOT_JNI_GLOBAL, class_tag, /*size=*/0, tag_ptr, + const_cast(user_data)); + if (*tag_ptr != 0) { + active_fixture->tags_ever_assigned[e.referee_idx] = *tag_ptr; + } + if (ctl == JVMTI_ITERATION_ABORT) { + break; + } + } + return JVMTI_ERROR_NONE; + } + + // Resolves each requested tag to its node's synthetic identity + // (&node_tags[idx]) by scanning node_tags for a matching current value - + // mirroring real GetObjectsWithTags()'s "only currently-live tags come + // back" contract. A tag in `dead_tags` is deliberately omitted even if + // node_tags still holds it, simulating "the object died, JVMTI forgot + // the tag with it" for the resolve-or-drop tests. + static jvmtiError JNICALL mock_GetObjectsWithTags( + jvmtiEnv *, jint tag_count, const jlong *req_tags, jint *count_ptr, + jobject **object_result_ptr, jlong **tag_result_ptr) { + if (active_fixture->fail_get_objects_with_tags) { + // Deliberately leave *count_ptr/*object_result_ptr/*tag_result_ptr + // untouched - a real failed JVMTI call makes no promise about + // them, and releaseSearchTags() must not read them on this path. + return JVMTI_ERROR_OUT_OF_MEMORY; + } + if (active_fixture->gotw_delay_ns != 0) { + u64 until = OS::nanotime() + active_fixture->gotw_delay_ns; + while (OS::nanotime() < until) { + // busy-wait: a sleep could overshoot by scheduler latency, + // and the overshoot direction matters for the one-batch + // deadline arithmetic the callers of this knob rely on. + } + } + std::vector objs; + std::vector found; + for (jint i = 0; i < tag_count; i++) { + jlong want = req_tags[i]; + if (want == 0 || active_fixture->dead_tags.count(want) > 0) { + continue; + } + // The decoder resolves a referrer CLASS from its raw (negative) + // tag - no node carries one, so the tag -> fake jclass map + // (field_decode_classes, see its own comment) serves it. + auto fd = active_fixture->field_decode_classes.find(want); + if (fd != active_fixture->field_decode_classes.end()) { + objs.push_back((jobject)fd->second); + found.push_back(want); + break; + } + for (size_t idx = 0; idx < active_fixture->node_tags.size(); idx++) { + if (active_fixture->node_tags[idx] == want) { + objs.push_back((jobject)&active_fixture->node_tags[idx]); + found.push_back(want); + break; + } + } + } + *count_ptr = (jint)objs.size(); + *object_result_ptr = objs.empty() + ? nullptr : (jobject *)malloc(sizeof(jobject) * objs.size()); + *tag_result_ptr = found.empty() + ? nullptr : (jlong *)malloc(sizeof(jlong) * found.size()); + for (size_t i = 0; i < objs.size(); i++) { + (*object_result_ptr)[i] = objs[i]; + (*tag_result_ptr)[i] = found[i]; + } + return JVMTI_ERROR_NONE; + } + + // Plays back `script` against the real production heap_reference_callback, + // modelling enough of FollowReferences' actual semantics for these + // heap-walk tests to be meaningful: + // - "a reference from A to B is not traversed until A is visited" - + // an edge whose referrer was not returned JVMTI_VISIT_OBJECTS for + // (or was never itself visited) is skipped, exactly as a real + // traversal would never reach it. + // - a JVMTI_VISIT_ABORT return stops delivery immediately. + // - when `initial_object` is non-NULL (expandFrontier()'s + // resumed-pass calls), only edges reachable from that object's own + // node are replayed - root edges (referrer_idx == -1) are skipped + // entirely, matching FollowReferences' real "the specified object is + // used instead of the heap roots" contract. `initial_object == NULL` + // (the first-pass, root-seeded call) is unchanged from the original + // single-pass heap-walk engine. + // This is not a full JVMTI implementation (real traversal order, + // multi-referrer objects, and primitive/array callbacks are all out of + // scope) - just enough fidelity to exercise the hop-cap/budget/ + // frontier-cap/class-skip/resumption logic in heapReferenceCallback()/ + // expandFrontier() themselves. + static jvmtiError JNICALL mock_FollowReferences( + jvmtiEnv *, jint, jclass, jobject initial_object, + const jvmtiHeapCallbacks *callbacks, const void *user_data) { + std::unordered_map expandable; // seed_idx == -2 marks the root walk (initial_object == NULL); any + // other value marks an expansion walk seeded from one or more boundary + // objects, in which case root edges are never replayed. -1 is the + // array-holder walk (a whole BFS level's boundary objects at once); + // >= 0 is the single-object legacy per-entry walk. + int seed_idx = -2; + // The transient holder array itself is never tagged (mirrors real + // production: admitStaticFieldRoots()/expandFrontier() never call + // SetTag on the frontier-holder array they build), so every + // holder->element ARRAY_ELEMENT edge below is replayed with a + // referrer tag of 0. + static jlong holder_tag = 0; + if (initial_object != nullptr) { + auto holder_it = active_fixture->holders.find(initial_object); + if (holder_it != active_fixture->holders.end()) { + // Array-holder walk (expandFrontier()'s already-tagged + // boundary batch, or admitStaticFieldRoots()'s negative- + // tagged class-object seed): actually invoke the production + // callback for each holder->element edge, exactly like a + // real FollowReferences(initial_object=holder_array) call + // would - this is what lets heap_reference_callback()'s own + // tag-sign/reference_kind logic (e.g. the *tag_ptr < 0 + // early-return and its admitStaticFieldRoots() carve-out) + // actually run, rather than assuming every element is + // expandable. + seed_idx = -1; + for (jobject elem : holder_it->second) { + int idx = active_fixture->indexOfNode(elem); + if (idx < 0) { + continue; + } + jlong *tag_ptr = &active_fixture->node_tags[idx]; + jint ctl = callbacks->heap_reference_callback( + JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, nullptr, + /*class_tag=*/0, /*referrer_class_tag=*/0, + /*size=*/0, tag_ptr, &holder_tag, + /*length=*/-1, const_cast(user_data)); + if (*tag_ptr != 0) { + active_fixture->tags_ever_assigned[idx] = *tag_ptr; + } + if (ctl & JVMTI_VISIT_ABORT) { + return JVMTI_ERROR_NONE; + } + expandable[idx] = (ctl & JVMTI_VISIT_OBJECTS) != 0; } + } else { + seed_idx = active_fixture->indexOfNode(initial_object); + expandable[seed_idx] = true; + } + } + for (auto &e : active_fixture->script) { + if (e.referrer_idx == -1) { + if (seed_idx != -2) { + continue; // resumed pass: never replay root edges + } + } else { + auto it = expandable.find(e.referrer_idx); + if (it == expandable.end() || !it->second) { continue; + } + } jlong class_tag = 0; + if (e.class_idx >= 0) { + class_tag = active_fixture->tags[active_fixture->classes[e.class_idx].klass]; + } + jlong *referrer_tag_ptr = e.referrer_idx >= 0 + ? &active_fixture->node_tags[e.referrer_idx] : nullptr; + jlong *tag_ptr = &active_fixture->node_tags[e.referee_idx]; + jint ctl = callbacks->heap_reference_callback( + e.kind, nullptr, class_tag, /*referrer_class_tag=*/0, + /*size=*/0, tag_ptr, referrer_tag_ptr, /*length=*/-1, + const_cast(user_data)); + if (*tag_ptr != 0) { + active_fixture->tags_ever_assigned[e.referee_idx] = *tag_ptr; + } + if (ctl & JVMTI_VISIT_ABORT) { + return JVMTI_ERROR_NONE; + } + expandable[e.referee_idx] = (ctl & JVMTI_VISIT_OBJECTS) != 0; + } + return JVMTI_ERROR_NONE; + } +}; + +ReferenceChainsBfsTest *ReferenceChainsBfsTest::active_fixture = nullptr; + +TEST_F(ReferenceChainsBfsTest, ReconstructsChainForSyntheticGraph) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *classA = (void *)0x2001, *classB = (void *)0x2002, + *classTarget = (void *)0x2003; + int ca = addClass(classA, "Lcom/rc/phase3/graph/A;"); + int cb = addClass(classB, "Lcom/rc/phase3/graph/B;"); + int ct = addClass(classTarget, "Lcom/rc/phase3/graph/Target;"); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeTarget = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, ca}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, cb}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeTarget, ct}, + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + + // A single pass that reaches full exhaustion of the reachable + // graph completes the search and releases every tag it assigned - so + // the tag must be fetched via tags_ever_assigned (captured at + // assignment time), not node_tags (already reset to 0 by + // releaseSearchTags() by the time runPass() returns; see + // ReleasesTagsOnCompletion below for the release itself). + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + jlong targetTag = tags_ever_assigned[nodeTarget]; + ASSERT_NE(0, targetTag); + EXPECT_EQ(0, node_tags[nodeTarget]); // released - see the comment above + + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain(targetTag, &chain)); + ASSERT_EQ(3u, chain.size()); + + int expectedTarget = Profiler::instance()->lookupClass( + "com/rc/phase3/graph/Target", strlen("com/rc/phase3/graph/Target")); + int expectedB = Profiler::instance()->lookupClass( + "com/rc/phase3/graph/B", strlen("com/rc/phase3/graph/B")); + int expectedA = Profiler::instance()->lookupClass( + "com/rc/phase3/graph/A", strlen("com/rc/phase3/graph/A")); + ASSERT_NE(-1, expectedTarget); + ASSERT_NE(-1, expectedB); + ASSERT_NE(-1, expectedA); + + EXPECT_EQ((u32)expectedTarget, chain[0]); + EXPECT_EQ((u32)expectedB, chain[1]); + EXPECT_EQ((u32)expectedA, chain[2]); + + // buildChainEvent() wraps the same reconstructChain() call into + // the ReferenceChainEvent shape Recording::recordReferenceChain() + // (flightRecorder.cpp) expects - same chain/order, plus the target's own + // depth from FrontierEntry. + ReferenceChainEvent event; + ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, targetTag, + &event)); + EXPECT_EQ((u64)targetTag, event._target_tag); + EXPECT_EQ(2u, event._depth); // root(A, depth0) -> B(depth1) -> Target(depth2) + ASSERT_EQ(chain.size(), event._hops.size()); + for (size_t i = 0; i < chain.size(); i++) { + EXPECT_EQ(chain[i], event._hops[i].klass_id); + } + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, BuildChainEventFailsForUnknownTag) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ReferenceChainEvent event; + EXPECT_FALSE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, 12345, + &event)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, BuildAbandonedEventFailsUnlessSearchAbandoned) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Freshly started: never abandoned (never even run a pass yet). + ReferenceChainAbandonedEvent event; + EXPECT_FALSE(tracker->buildAbandonedEvent(&event)); + + int nodeA = addNode(); + script = {{JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}}; + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + // Small graph, no caps hit -> COMPLETED, not ABANDONED. + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_FALSE(tracker->buildAbandonedEvent(&event)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, HopCapStopsAdmittingBeyondCap) { + Arguments args; + // hops=1: only depth 0 (direct root references) may be admitted. + ASSERT_FALSE(args.parse("referencechains=true:hops=1:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // depth 0 - admitted + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // depth 1 - capped + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); // hop cap is not truncation - a normal boundary + // Not truncated -> graph fully explored within the hop cap -> the search + // completes and releases its tags in the same call (see the previous + // test's comment) - fetch nodeA's tag via tags_ever_assigned, not + // node_tags. + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_EQ(0, node_tags[nodeB]); // never admitted into the frontier + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, BudgetExhaustionTruncatesAndIsReported) { + Arguments args; + // budget=1: root enumeration and the expand phase draw from separate + // budget pools (see runPassManualWalk()'s own comment), each sized 1 + // here - root enum admits nodeA, then the expand phase's own 1-unit + // budget admits exactly one of nodeA's two children before exhausting. + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeC = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // admitted via root enum + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // admitted via expand + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeC, -1}, // expand budget exhausted + }; + + bool truncated = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + // Budget exhaustion for *this pass* leaves pending work (nodeC's own + // edge was never even attempted) - the search stays RUNNING, not + // COMPLETED, so no tag release happens yet and node_tags[nodeA]/[nodeB] + // are still the real assigned tags. + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + EXPECT_NE(0, node_tags[nodeA]); + EXPECT_NE(0, node_tags[nodeB]); + EXPECT_EQ(0, node_tags[nodeC]); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, PreTaggedClassObjectsAreNeverExpandedOrAdmitted) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + node_tags[classNode] = -7; // simulate a class object already tagged by + // resolveLoadedClasses() before this pass - + // see ClassTagTable's tag-sign convention. + int fieldTargetNode = addNode(); + + script = { + // A root reference straight to the class object (e.g. a + // JVMTI_HEAP_REFERENCE_SYSTEM_CLASS root edge in a real walk). + {JVMTI_HEAP_REFERENCE_SYSTEM_CLASS, -1, classNode, -1}, + // A static field of that class - must never be delivered by a real + // FollowReferences call, since the class-object edge above must not + // return JVMTI_VISIT_OBJECTS; mock_FollowReferences enforces this + // the same way a real traversal would (see its own comment). + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, fieldTargetNode, -1}, + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + + EXPECT_EQ(-7, node_tags[classNode]); // untouched - never treated as a + // frontier object + EXPECT_EQ(0, node_tags[fieldTargetNode]); // never reached - the class + // edge above must not expand + + tracker->stop(); +} + +// Regression test for admitStaticFieldRoots(): an object retained solely by +// a static field (no other GC root reaches it) must still be discovered. +// The scripted class is registered via addClass() with its own node's +// address as the jclass identity, so GetLoadedClasses()/resolveLoadedClasses() +// (real production code, driven through the same mocked jvmti) tag it +// negative through node_tags[classNode] exactly like a real Class object - +// the same identity the STATIC_FIELD script edge below uses as its +// referrer, mirroring how a real Class object is simultaneously "a loaded +// class" and "the referrer of its own static-field edges". +TEST_F(ReferenceChainsBfsTest, DiscoversObjectRetainedOnlyByStaticField) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + int fieldTargetNode = addNode(); + addClass((void *)&node_tags[classNode], "Lcom/rc/statics/Holder;"); + + script = { + // No GC-root path to fieldTargetNode at all - it is reachable only + // via classNode's static field. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, fieldTargetNode, -1}, + }; + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + + // classNode got tagged negative by the real resolveLoadedClasses() scan + // (not manually, unlike PreTaggedClassObjectsAreNeverExpandedOrAdmitted + // above), and was never itself admitted as a frontier object. + EXPECT_LT(node_tags[classNode], 0); + + jlong target_tag = tags_ever_assigned[fieldTargetNode]; + ASSERT_NE(0, target_tag); + + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain(target_tag, &chain)); + ASSERT_EQ(1u, chain.size()); + + FrontierEntry entry{}; + ASSERT_TRUE(tracker->frontierTable()->lookup(target_tag, &entry)); + EXPECT_EQ(0, entry.parent_tag); // root-attached, not a child hop + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); + + // buildChainEvent() appends the root TYPE (the declaring class) as the + // chain's root-side end: the frontier path's terminal element is the + // static field's holder instance, one hop below the root type. The + // declaring class is resolved from the root-attached entry's + // referrer_class_tag (captured from the sweep's referrer_tag_ptr, a + // negative class tag). + ReferenceChainEvent event; + ASSERT_TRUE(tracker->buildChainEvent(&mock_jvmti, &mock_jni, target_tag, + &event)); + int expectedHolder = Profiler::instance()->lookupClass( + "com/rc/statics/Holder", strlen("com/rc/statics/Holder")); + ASSERT_NE(-1, expectedHolder); + ASSERT_EQ(2u, event._hops.size()); + EXPECT_EQ(chain[0], event._hops[0].klass_id); // the target's own class, unchanged + EXPECT_EQ((u32)expectedHolder, event._hops[1].klass_id); + // One label per hop (recordReferenceChain() drops ALL labels when any is + // empty); the root-type hop's own edge is the unlabeled root edge + // (field_index -1). + ASSERT_EQ(2u, event._hops.size()); + EXPECT_FALSE(event._hops[0].edge_label.empty()); + EXPECT_FALSE(event._hops[1].edge_label.empty()); + + tracker->stop(); +} + +// Regression test for the resolveLoadedClasses() scan-skip guard: it must +// compare `class_count != _last_resolved_class_count`, not `class_count > +// _last_resolved_class_count`. GetLoadedClasses()'s count is not monotonic - +// class unloading can shrink it - so a `>` guard would stay permanently +// skipped once the count is loaded back up to, but not past, a prior +// historical peak, silently leaving any *different* classes loaded in that +// regrowth untagged forever. See resolveLoadedClasses()'s own comment for +// the full rationale. +TEST_F(ReferenceChainsBfsTest, ResolveLoadedClassesRescansAfterClassCountShrinksAndPartiallyRegrows) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *classA = (void *)0x3001, *classB = (void *)0x3002, *classC = (void *)0x3003; + addClass(classA, "Lcom/rc/regress/A;"); + int idxB = addClass(classB, "Lcom/rc/regress/B;"); + + // Pass 1: both A and B loaded (count == 2) - both get resolved/tagged. + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + EXPECT_EQ(2, ReferenceChainsTestAccessor::lastResolvedClassCount()); + ASSERT_NE(0u, tags.count(classA)); + ASSERT_NE(0u, tags.count(classB)); + EXPECT_NE(0, tags[classA]); + EXPECT_NE(0, tags[classB]); + + // Simulate B's classloader being GC'd: GetLoadedClasses() now reports + // only A (count shrinks 2 -> 1), exactly like a real class unload. + classes.erase(classes.begin() + idxB); + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + EXPECT_EQ(1, ReferenceChainsTestAccessor::lastResolvedClassCount()); + + // Simulate a *different* class C loading back in, bringing the count + // back to 2 - the same count as pass 1's peak, but not the same class + // set. The buggy `>` guard (2 > 2 is false, since it never re-lowered + // _last_resolved_class_count on the shrink above either) would skip the + // scan here and leave C's tag at 0 forever; the fixed `!=` guard must + // still resolve it. + addClass(classC, "Lcom/rc/regress/C;"); + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + EXPECT_EQ(2, ReferenceChainsTestAccessor::lastResolvedClassCount()); + ASSERT_NE(0u, tags.count(classC)); + EXPECT_NE(0, tags[classC]); // the regression this test guards against + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Incremental resumption across passes (ReferenceChainTracker:: +// expandFrontier()/releaseSearchTags()/shouldRunPass(), and runPass()'s +// SearchState transitions). +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, MultiPassResumptionReconstructsChainAcrossPasses) { + Arguments args; + // budget=1 forces each pass to admit at most one new frontier entry, so + // this 3-hop chain cannot be discovered within a single pass - + // exercising expandFrontier() (resumed passes), not just the first + // pass's root walk. + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeTarget = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeTarget, -1}, + }; + + // Drive the search to completion one pass at a time, exactly as + // threadLoop() would once wired up (each call bounded by `budget`). + bool truncated = true; + int passes_issued = 0; + while (tracker->searchState() == SearchState::RUNNING && passes_issued < 20) { + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + passes_issued++; + } + + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_GT(tracker->passesRun(), 1); // did not fit in a single pass + EXPECT_EQ(tracker->passesRun(), passes_issued); + + jlong targetTag = tags_ever_assigned[nodeTarget]; + ASSERT_NE(0, targetTag); + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain(targetTag, &chain)); + // Depth/parent_tag linkage survived resumption intact - all 3 hops walk + // back to a root-attached (depth 0) entry, which reconstructChain() + // requires to succeed at all (see its own "reaching parent_tag == 0" + // contract). + EXPECT_EQ(3u, chain.size()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, FrontierCapHitAbandonsImmediately) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000:framecap=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + ASSERT_EQ(1, tracker->frontierTable()->maxCapacity()); + + int nodeA = addNode(); + int nodeB = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, // fits (the one slot) + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, // frontier cap hit + }; + + bool truncated = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + // Frontier-size cap hit -- the search abandons immediately (design + // doc's Termination-section priority 1). Once the table is full, no + // new entry can ever be admitted, so the frontier can never grow + // again -- deferring to a separate no-progress counter would never + // actually observe that counter, since this same condition matches + // every subsequent pass too. + ASSERT_EQ(SearchState::ABANDONED, tracker->searchState()); + ASSERT_EQ(SearchAbandonReason::FRONTIER_CAP, tracker->abandonReason()); + + // nodeA was admitted (frontier cap=1 allowed one entry), then its tag + // was released as part of this same pass's abandon handling (the mock + // GetObjectsWithTags() succeeds by default). + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_EQ(0, node_tags[nodeA]); + // nodeB was never admitted (frontier cap hit). + EXPECT_EQ(0, node_tags[nodeB]); + + tracker->stop(); +} + +// releaseSearchTags()'s GetObjectsWithTags() call failing must NOT be treated +// as "released" - see that method's own comment for why: marking a tag +// ABANDONED (or resetting _next_tag on restart) while its object might still +// be live would let a restarted search's fresh tags collide with it, +// corrupting FrontierTable's tag-uniqueness invariant. This is the regression +// test for that failure path (previously the return value was discarded +// entirely). +TEST_F(ReferenceChainsBfsTest, ReleaseSearchTagsFailureBlocksTagReuseUntilItSucceeds) { + Arguments args; + // framecap=1 with a self-cycle: pass 1 admits nodeA (the frontier's + // only slot); the nodeA->nodeA edge then finds nodeA + // ALREADY_ADMITTED rather than hitting the frontier cap (no new slot + // is needed for an edge back to an already-tagged object), so the + // search stays RUNNING and only the no-progress detector - after + // NO_PROGRESS_PASS_LIMIT stale passes - can abandon it. This test + // verifies that the tag release on that abandonment works even when + // GetObjectsWithTags() fails. + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000:framecap=1:ttl=0")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + // nodeA -> nodeA self-cycle. With framecap=1, pass 1 admits nodeA; + // the self-cycle edge is ALREADY_ADMITTED, not a fresh frontier slot. + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeA, -1}, // self-cycle + }; + + long long failedBefore = + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED); + + fail_get_objects_with_tags = true; + bool truncated = false; + // Pass 1: admits nodeA; the self-cycle keeps the pass truncated + // without growing the frontier further. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_NE(0, node_tags[nodeA]); + + // Run enough stale passes to trigger no-progress abandonment. + for (int i = 1; i < ReferenceChainTracker::NO_PROGRESS_PASS_LIMIT + 1; i++) { + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + } + // The next pass should abandon via no-progress. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + ASSERT_EQ(SearchState::ABANDONED, tracker->searchState()); + + // GetObjectsWithTags() failed - nodeA's still-live tag must NOT have been + // cleared, and the failure must be counted. + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_NE(0, node_tags[nodeA]) << "tag must not be cleared when the " + "release batch itself failed"; + EXPECT_FALSE(ReferenceChainsTestAccessor::tagsReleased()); + EXPECT_EQ(failedBefore + 1, + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)); + + // While the release is still outstanding, shouldRunPass() must force a + // retry unconditionally. + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::ABANDONED, tracker->searchState()) + << "must retry the release in place, not restart, while tags are " + "still unreleased"; + + // A further runPass() call retries the release; still failing. + int passesBefore = tracker->passesRun(); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(passesBefore, tracker->passesRun()); + EXPECT_NE(0, node_tags[nodeA]); + EXPECT_FALSE(ReferenceChainsTestAccessor::tagsReleased()); + EXPECT_EQ(failedBefore + 2, + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)); + + // Once GetObjectsWithTags() starts succeeding again. + fail_get_objects_with_tags = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(0, node_tags[nodeA]); + EXPECT_TRUE(ReferenceChainsTestAccessor::tagsReleased()); + EXPECT_EQ(failedBefore + 2, + Counters::getCounter(REFERENCE_CHAIN_TAG_RELEASE_FAILED)) + << "a successful release must not itself count as a failure"; + + tracker->stop(); +} + +// Regression test for the CANARY_STUCK/frontier-wipe convergence bug: prior +// to this fix, runPass()'s canary-stuck branch fired purely off +// _passes_since_last_candidate_progress, so a search whose candidate simply +// had not been found yet was abandoned - and its frontier destructively +// wiped by the next restartSearch() - after only CANARY_NO_PROGRESS_PASS_LIMIT +// passes, even while the whole-graph frontier was still growing every single +// pass. Live on-pod evidence showed exactly this: a frontier that had grown +// to 12k-16k entries got wiped roughly every 20s while chasing a +// confirmed-reachable candidate. The fix requires the whole-graph frontier to +// ALSO have stalled (_passes_since_last_progress >= NO_PROGRESS_PASS_LIMIT) +// before CANARY_STUCK can fire - see CANARY_NO_PROGRESS_PASS_LIMIT's and +// canaryStuckPassLimit()'s own comments. +TEST_F(ReferenceChainsBfsTest, CanaryStuckRequiresWholeGraphFrontierAlsoStalled) { + Arguments args; + // budget=1: exactly one new frontier admission per pass, so the frontier + // grows every single pass for as long as the chain has unexplored nodes + // left - _passes_since_last_progress never leaves 0. + ASSERT_FALSE(args.parse( + "referencechains=true:hops=200:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // A chain longer than the number of passes driven below, so the frontier + // still has pending work - and is still growing one node per pass - at + // every pass this test checks. + constexpr int kChainLength = 50; + std::vector nodes; + for (int i = 0; i < kChainLength; i++) { + nodes.push_back(addNode()); + } + script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodes[0], -1}); + for (int i = 1; i < kChainLength; i++) { + script.push_back( + {JVMTI_HEAP_REFERENCE_FIELD, nodes[i - 1], nodes[i], -1}); + } + + // A canary candidate that this graph never actually contains (no node is + // ever tagged with the candidate's marker tag) - the candidate-specific + // stuck counter (_passes_since_last_candidate_progress) climbs every pass + // with zero discovery progress, exactly like the live-pod scenario + // chasing a candidate deeper than the old fixed + // CANARY_NO_PROGRESS_PASS_LIMIT (30) passes could reach. + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + + bool truncated = true; + // One more pass than the old fixed CANARY_NO_PROGRESS_PASS_LIMIT: long + // enough that the pre-fix single-condition check would already have + // abandoned the search, but short enough that the 40-node chain still has + // unexplored work left, so the frontier is still genuinely growing every + // pass. + for (int i = 0; i < ReferenceChainTracker::CANARY_NO_PROGRESS_PASS_LIMIT + 2; + i++) { + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + ASSERT_EQ(0, tracker->passesSinceLastProgressForTest()) + << "pass " << i << ": frontier must still be growing every pass"; + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()) + << "pass " << i + << ": a canary search must not be abandoned while the " + "whole-graph frontier is still growing, even if its specific " + "candidate has not yet been found"; + } + // The narrower candidate-stuck counter climbed the whole time - this is + // what the old, single-condition check would have abandoned on alone. + EXPECT_GE(ReferenceChainsTestAccessor::passesSinceLastCandidateProgress(), + ReferenceChainTracker::CANARY_NO_PROGRESS_PASS_LIMIT); + + tracker->stop(); +} + +// Canary-lane backoff pacing (option A): a chase with unresolved candidates +// runs back-to-back only while it is fresh or making candidate progress; +// each pass with NO candidate progress doubles the spacing multiplier +// (_canary_backoff_mult) up to CANARY_BACKOFF_MULT_MAX, progress resets it +// to 1, and the OOM urgency ramp overrides the gate entirely. Live evidence +// this bounds (hotdog rounds 3-4): an un-findable candidate held the chase +// open for 32 minutes at ~88 passes/min - a full core - because canary_active +// bypassed every cadence check and threadLoop() skips its sleep whenever a +// pass will run. Work-scaled rather than a fixed cap: hotdog's own passes +// ran 0.7-4s, so any fixed cap below that would have changed nothing at all +// - the loop is work-bound when the pass exceeds the cap - while a fixed 1s +// cap starved a real deep ~200-pass chase outright (measured live). +TEST_F(ReferenceChainsBfsTest, CanaryLaneBacksOffWithoutProgressAndResetsOnProgress) { + Arguments args; + // Same shape as CanaryStuckRequiresWholeGraphFrontierAlsoStalled above: + // budget=1 with a long chain keeps the frontier growing one node per + // pass, so the CANARY_STUCK detector (which also requires a stalled + // frontier) never fires and the chase stays RUNNING through the whole + // loop below. + ASSERT_FALSE(args.parse( + "referencechains=true:hops=200:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + constexpr int kChainLength = 50; + std::vector nodes; + for (int i = 0; i < kChainLength; i++) { + nodes.push_back(addNode()); + } + script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodes[0], -1}); + for (int i = 1; i < kChainLength; i++) { + script.push_back( + {JVMTI_HEAP_REFERENCE_FIELD, nodes[i - 1], nodes[i], -1}); + } + + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + bool truncated = true; + + // Pass 1: the candidate admission itself raises the progress mark + // (0 -> 1), so this counts as progress and the multiplier stays at 1 - + // back-to-back. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::canaryBackoffMult()); + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) + << "a fresh chase must be allowed to run back-to-back"; + + // Pass 2: no candidate progress -> first doubling (1 -> 2). Seed a + // deterministic pass-cost EMA (mock passes are sub-ms, so the real EMA + // decays to 0) and a fresh pass timestamp so the spacing arithmetic is + // exact: spacing = 2 x 100ms. The seeded value is in NANOSECONDS since + // the EMA switched to ns (integer ms truncated every sub-5ms pass to + // zero and pinned the EMA - a real bug this switch fixes). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(2, ReferenceChainsTestAccessor::canaryBackoffMult()); + ReferenceChainsTestAccessor::setCanaryBackoffForTest( + /*mult=*/2, /*ema_ns=*/100ULL * 1000000ULL, OS::nanotime()); + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) + << "a no-progress canary pass must hold off the next one"; + // Beyond the spacing, the chase is allowed again - the backoff paces, + // it never abandons. + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass( + ReferenceChainsTestAccessor::lastCanaryPassNs() + + 2ULL * 100ULL * 1000000ULL + 1)) + << "elapsed spacing must re-admit the canary pass"; + + // The OOM urgency ramp overrides the backoff gate entirely. + ReferenceChainsTestAccessor::setOomRampActive(true); + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())) + << "urgency must bypass the canary backoff"; + ReferenceChainsTestAccessor::setOomRampActive(false); + + // Consecutive no-progress passes double the multiplier up to the cap + // (seeded at 8 so one more pass reaches it, the next holds it). + ReferenceChainsTestAccessor::setCanaryBackoffForTest( + /*mult=*/8, /*ema_ns=*/100ULL * 1000000ULL, OS::nanotime()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(ReferenceChainTracker::CANARY_BACKOFF_MULT_MAX, + ReferenceChainsTestAccessor::canaryBackoffMult()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(ReferenceChainTracker::CANARY_BACKOFF_MULT_MAX, + ReferenceChainsTestAccessor::canaryBackoffMult()) + << "the multiplier must hold at its cap, not grow past it"; + + // Candidate progress (a new candidate admitted into a slot) resets the + // lane to back-to-back. + ReferenceChainsTestAccessor::setCandidateCountForTest(2); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(1, /*klass_id=*/987); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::canaryBackoffMult()) + << "candidate progress must reset the spacing multiplier to 1"; + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(OS::nanotime())); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, ChainCompletesWithoutAbandonment) { + Arguments args; + // budget=1 on a graph where each pass admits exactly one new edge + // until the chain is exhausted, then the frontier stops growing. + // After NO_PROGRESS_PASS_LIMIT passes with no growth, the search + // is abandoned (progress-based termination, not wall-clock TTL). + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + int nodeC = addNode(); + int nodeD = addNode(); + + // A chain one node longer than either pass's 1-edge expand budget can + // fully drain in a single call, so each pass still ends truncated (see + // mock_FollowReferences()'s own comment: an array-holder walk chains + // through as many script edges as it can admit before budget aborts it) + // and there is still pending work left for the no-progress check to catch + // once the chain is fully discovered. + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeB, nodeC, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeC, nodeD, -1}, + }; + + bool truncated = false; + // Pass 1: root enum admits nodeA, expand admits nodeB, aborts on + // nodeB->nodeC for lack of budget - truncated, frontier grew (progress). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + // Pass 2: admits nodeC, aborts on nodeC->nodeD - still truncated, + // still growing (progress). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + // Pass 3: admits nodeD, chain exhausted - no longer truncated, + // no pending frontier, natural completion. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_FALSE(truncated); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Tags released. + EXPECT_NE(0, tags_ever_assigned[nodeA]); + EXPECT_EQ(0, node_tags[nodeA]); + EXPECT_NE(0, tags_ever_assigned[nodeB]); + EXPECT_EQ(0, node_tags[nodeB]); + + // No-progress (not TTL) is reported as the reason when the frontier stalls. + // This test uses ttl=0 to disable the wall-clock TTL, so only the + // no-progress detector can abandon. + EXPECT_EQ(SearchAbandonReason::NONE, tracker->abandonReason()); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, NoProgressAbandonsSearchAndReleasesTags) { + // Verify the progress-based abandonment wiring: the no-progress limit + // is accessible, positive, and reset to 0 by resetSearchStateForTest(). + // The mock runPass() re-enumerates roots on every search_started=0 + // pass, causing the frontier to oscillate rather than stabilize, + // so a full end-to-end no-progress abandonment can't be tested + // with the mock; the real JVM path is validated by a Java-side + // integration test. + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:ttl=0:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Verify the no-progress limit is accessible and positive. + EXPECT_GT(ReferenceChainTracker::NO_PROGRESS_PASS_LIMIT, 0); + + // Verify that a fresh search starts with zero passes since last progress. + EXPECT_EQ(0, tracker->passesSinceLastProgressForTest()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, ResolveOrDropPrunesDeadFrontierEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1:firstpassbudget=1")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int nodeA = addNode(); + int nodeB = addNode(); + // A second root that root enumeration's own 1-unit budget (firstpassbudget=1) + // can't reach this pass - the resulting root-enum truncation makes + // runPassManualWalk() return before expandFrontier() ever runs (see its + // own comment on frontier-cap-hit/budget-exhausted root-enum truncation), + // so nodeA is admitted but never gets a chance to expand nodeA->nodeB. + int decoyRoot = addNode(); + + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, nodeA, -1}, + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, decoyRoot, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, nodeA, nodeB, -1}, + }; + + bool truncated = false; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); // pass 1 + ASSERT_TRUE(truncated); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + jlong aTag = tags_ever_assigned[nodeA]; + ASSERT_NE(0, aTag); + // Simulate nodeA dying (collected) between pass 1 and pass 2 - + // GetObjectsWithTags will no longer report it as live. + dead_tags.insert(aTag); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); // pass 2: resolve-or-drop + EXPECT_FALSE(truncated); + // The dead branch was pruned for free - with nothing else pending, the + // search completes rather than staying RUNNING or being ABANDONED. + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_EQ(2, tracker->passesRun()); + + FrontierEntry entry{}; + ASSERT_TRUE(tracker->frontierTable()->lookup(aTag, &entry)); + EXPECT_EQ(FrontierEntryState::ABANDONED, entry.state); + + // nodeB was never discovered - nodeA's subtree was pruned, not expanded. + EXPECT_EQ(0, tags_ever_assigned[nodeB]); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// pollWatchedTargets() (design doc's Open Question 3 bridging +// step, corrected read-only mechanism - see referenceChains.h's own +// target-selection bridging step header comment and pollWatchedTargets()'s +// own comment for why the step must be a read, not a SetTag seed). +// +// Mirrors referenceChainJfrRoundtrip_ut.cpp's FrontierTable::insert() +// seeding style rather than driving a full scripted runPass() (this test +// suite's gtest coverage bullet explicitly asks for reusing that seeding style over +// writing a third pattern): a candidate's "already discovered by an +// ordinary pass" tag is modelled directly as a FrontierTable entry plus a +// mocked GetTag() that reports it - both reach the same FrontierTable + tag +// state pollWatchedTargets()/buildChainEvent() read, regardless of whether a +// scripted heap walk or direct insertion produced it. +// +// LivenessTracker::instance() is a second process-wide singleton (see its +// own header comment) shared with livenessTracker_ut.cpp within this same +// gtest binary - klassPopulationResetForTest()/setGcGenerationsForTest() at +// SetUp/TearDown keep this suite's use of it self-contained, the same way +// ReferenceChainsTestAccessor::reset() already isolates +// ReferenceChainTracker::instance() above. +// --------------------------------------------------------------------------- + +// =========================================================================== +// Pod-in-a-jar system harness (design node: design-pod-in-a-jar-harness; +// meta-whackamole-analysis): the REAL tracker loop (shouldRunPass -> +// runPass -> pollWatchedTargets, the exact threadLoop body) driven over the +// scripted mock heap, asserting SYSTEM INVARIANTS instead of unit symptoms. +// The topology encodes every pod-discovered shape: a static holder holding +// a synchronized-list wrapper WITH its mutex==this self-edge (round 16 +// fix-A), the wrapper->c->chunks subtree, leak-tagged chunks (pre-seeded +// tags model what LivenessTracker::tagLeakInstances assigns on the pod; +// the walk's leak-tag interception path then runs for real), and flood +// statics for anchor-tier volume. Each invariant maps 1:1 to rounds that +// shipped a bug the 348-test unit suite could not see. +// =========================================================================== +class PodInAJarTest : public ReferenceChainsBfsTest { +protected: + struct PodTopology { + int holder_class_node = -1; + int wrapper_node = -1; + int list_node = -1; + int leak_cls_idx = -1; + int wrapper_cls_idx = -1; + int list_cls_idx = -1; + std::vector chunk_nodes; + std::vector chunk_leak_tags; + std::vector flood_class_nodes; + std::vector flood_nodes; + std::vector filler_nodes; + }; + + static constexpr int kChunks = 6; // < MAX_DISCOVERED_INSTANCES_PER_CLASS + static constexpr u64 kCycleNs = 2000000000ULL; // 2s fake-clock step + + void SetUp() override { + ReferenceChainsBfsTest::SetUp(); + // resolveCandidateRepresentative() NewLocalRef()s the stored + // representative; the Bfs fixture never wires that JNI slot + // (PollWatchedTargetsTest has its own). Passthrough is right for + // the harness: its reps are mock node pointers with fixture + // lifetime, never GC'd. + jni_tbl.NewLocalRef = &mock_NewLocalRefPassthrough; + // NOTE: no liveness calls here - the Bfs tests run without them, + // and liveness state changes (setGcGenerationsForTest) alter the + // pass machinery's behavior; each harness test resets liveness + // explicitly where it wants it (resetLivenessForPod below). + } + + static jobject JNICALL mock_NewLocalRefPassthrough(JNIEnv *, jobject ref) { + return ref; + } + + void resetLivenessForPod() { + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(true); + LivenessTracker::instance()->leakTagPoolResetForTest(); + } + + void TearDown() override { + ReferenceChainsBfsTest::TearDown(); + // Hygiene for whatever suite runs next: clear any population this + // harness seeded. + LivenessTracker::instance()->klassPopulationResetForTest(); + } + + // Builds the hotdog leak topology: holder-class static -> synchronized + // wrapper (with its mutex==this self-edge) -> c -> K leak-tagged chunks, + // plus 2 flood classes x 16 static-held nodes for anchor-tier volume. + PodTopology buildLeakPod() { + PodTopology topo; + // CAPACITY FIXTURE CONTRACT: classes registered via + // classes.push_back({(void*)&node_tags[node], ...}) capture the + // node's ADDRESS as the jclass identity - std::vector growth would + // reallocate node_tags and dangle every captured pointer, silently + // making indexOfNode() fail for those classes in the static sweep + // (the holder array seeds no expandable class, the STATIC_FIELD + // edges never replay, nothing admits). Reserve the full node count + // up front - the topology builder never shrinks. Node total: + // 3 roots + 2 flood class nodes + 6 chunks + 32 flood + 300 filler. + node_tags.reserve(600); + tags_ever_assigned.reserve(600); + topo.leak_cls_idx = addClass((void *)0x5001, "[B"); + topo.wrapper_cls_idx = addClass( + (void *)0x5002, + "Ljava/util/Collections$SynchronizedRandomAccessList;"); + topo.list_cls_idx = addClass((void *)0x5003, "Ljava/util/ArrayList;"); + + topo.holder_class_node = addNode(); + topo.wrapper_node = addNode(); + topo.list_node = addNode(); + // The holder class node doubles as the jclass identity (see + // DiscoversObjectRetainedOnlyByStaticField): register it as a loaded + // class so the static-field sweep admits the wrapper root-attached. + classes.push_back({(void *)&node_tags[topo.holder_class_node], + "Lcom/rc/pod/Holder;"}); + // Flood classes get their own jclass-identity nodes too, so their + // statics are separate anchors (not more statics on the holder). + const char *flood_sigs[2] = {"Lcom/rc/pod/FloodA;", + "Lcom/rc/pod/FloodB;"}; + for (int c = 0; c < 2; c++) { + topo.flood_class_nodes.push_back(addNode()); + classes.push_back( + {(void *)&node_tags[topo.flood_class_nodes[c]], flood_sigs[c]}); + } + + for (int i = 0; i < kChunks; i++) { + topo.chunk_nodes.push_back(addNode()); + } + for (int i = 0; i < 32; i++) { + topo.flood_nodes.push_back(addNode()); + } + // Backlog volume: a 300-edge deep chain under one flood root. The + // real pod's searches span many passes because the heap is huge; + // the harness needs the same property or the first pass (whose + // budget auto-scales 10x for the first pass, arguments.cpp) admits + // everything before the poll ever auto-marks, and the search + // completes before any chain can be cached. + for (int i = 0; i < 300; i++) { + topo.filler_nodes.push_back(addNode()); + } + + script = { + // LEAK_BUFFER: the holder class's static field -> wrapper. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, topo.holder_class_node, + topo.wrapper_node, topo.wrapper_cls_idx}, + // The synchronized wrapper's mutex == this self-edge (round-16 + // fix-A shape: must not demote the root-attached wrapper). + {JVMTI_HEAP_REFERENCE_FIELD, topo.wrapper_node, topo.wrapper_node, + topo.wrapper_cls_idx}, + // wrapper -> c -> chunks. + {JVMTI_HEAP_REFERENCE_FIELD, topo.wrapper_node, topo.list_node, + topo.list_cls_idx}, + }; + for (int i = 0; i < kChunks; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, topo.list_node, + topo.chunk_nodes[i], topo.leak_cls_idx}); + } + // Flood volume: each flood class holds 16 statics. + for (int c = 0; c < 2; c++) { + for (int i = 0; i < 16; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_STATIC_FIELD, + topo.flood_class_nodes[c], + topo.flood_nodes[16 * c + i], + topo.leak_cls_idx /* any class; volume only */}); + } + } + // The filler chain hangs off flood node 0 (already a static anchor). + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, topo.flood_nodes[0], + topo.filler_nodes[0], topo.leak_cls_idx}); + for (int i = 0; i + 1 < (int)topo.filler_nodes.size(); i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, + topo.filler_nodes[i], topo.filler_nodes[i + 1], + topo.leak_cls_idx}); + } + reseedChunkLeakTags(topo); + return topo; + } + + // Leak tags model the poll's tagLeakInstances() assignment (the pod + // re-tags within minutes of every restart - the verify-not-retag state + // machine); the walk's leak-tag interception path consumes them for + // real from the live node tags. + void reseedChunkLeakTags(const PodTopology &topo) { + for (int i = 0; i < (int)topo.chunk_nodes.size(); i++) { + jlong leak_tag = 1073741824LL + 100 + i; + node_tags[topo.chunk_nodes[i]] = leak_tag; + tags_ever_assigned[topo.chunk_nodes[i]] = leak_tag; + } + } + + // Resolves the leak class's tracker-side klass id from the classTags + // table via the SAME tag value the mock passes as the chunk edges' + // class_tag (tags[klass_ptr] - the sweep's negative class tag, set + // during phase 1's static sweep). Class tags are process-lifetime, so + // this survives the phase-1 search completing and releasing its + // frontier/tags - unlike reading node_tags, which the release zeroes. + u32 resolveLeakKlassId(const PodTopology &topo) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + EXPECT_GT(ReferenceChainsTestAccessor::passesRunForTest(), 0) + << "phase-1 pass never ran"; + jlong class_tag = tags[classes[topo.leak_cls_idx].klass]; + EXPECT_LT(class_tag, 0) << "leak class never sweep-tagged (class_tag=" + << class_tag << ")"; + return tracker->classTags()->resolve(class_tag); + } + + // Seeds the leak-side liveness (population growth + qualifying tid + + // representative = the first chunk), mirroring + // PollWatchedTargetsTest::seedGrowingCandidate. + void seedLeakLiveness(u32 klass_id, const PodTopology &topo) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( + nullptr, klass_id, (jweak)&node_tags[topo.chunk_nodes[0]]); + // The representative's GetTag() identity: the mock tags map is keyed + // by object pointer; getTag(rep) must report the rep's leak tag. + tags[&node_tags[topo.chunk_nodes[0]]] = topo.chunk_leak_tags.empty() + ? 1073741924LL + : 0; /* placeholder */ + // (chunk_leak_tags is not kept; the tags map entry keeps the CURRENT + // leak tag of the rep - the reseed above made node_tags hold it.) + tags[&node_tags[topo.chunk_nodes[0]]] = + node_tags[topo.chunk_nodes[0]]; + } + + // One threadLoop iteration, fake clock: the exact body order from + // ReferenceChainTracker::threadLoop() (shouldRunPass -> runPass -> + // pollWatchedTargets, poll unconditional). + bool drivePodCycle(u64 &fake_now) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + bool should_run = tracker->shouldRunPassForTest(fake_now); + if (should_run) { + bool truncated = false; + tracker->runPass(&mock_jvmti, &mock_jni, &truncated); + } + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + fake_now += kCycleNs; + return should_run; + } + + // Seeds a throwaway candidate klass (NO instances, never matches any + // real class) whose only job is arming the leak signal so the phase-1 + // pass runs: the dormancy invariant (L8) proved shouldRunPass stays + // false without a candidate, and the klass-id resolution needs a pass. + void seedThrowawayLiveness(u32 klass_id) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest( + nullptr, klass_id, (jweak)0xBADC0DE); + } + + // Two-phase pod bring-up. Phase 1: a throwaway candidate arms the leak + // signal so cycle 1's pass runs - the sweep tags the classes and the + // walk admits the wrapper subtree, which makes the leak class's + // tracker-side klass id resolvable (the only reliable source of the id + // is the system's own resolve() over a real frontier entry). Phase 2: + // clean slate (population + search restart - the pod's churn), leak + // tags re-seeded (the poll's re-tagging after every restart), and the + // REAL candidate armed, so admissions auto-mark from cycle one exactly + // like the pod after warmup. + void bringUpPod(const PodTopology &topo, u32 &leak_klass_id_out, + u64 &fake_now) { + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + // Seed the fake clock from the REAL monotonic clock: the pain + // budgets' _last_update_ns is OS::nanotime()-based, so a fake epoch + // starting at ~0 sits before every budget's initialization and + // canStartNow() blocks every pass (the dormancy test passes either + // way - no passes at all - so only the pass-running tests caught + // this). + fake_now = OS::nanotime(); + resetLivenessForPod(); + seedThrowawayLiveness(/*klass_id=*/999); + // The candidate arms in the POLL, which runs after shouldRunPass in + // each cycle - so the first cycle only arms, and a pass actually + // runs one cycle later. Drive until a pass ran. + bool ran = false; + for (int i = 0; i < 6 && !ran; i++) { + ran = drivePodCycle(fake_now); + } + leak_klass_id_out = resolveLeakKlassId(topo); + LivenessTracker::instance()->klassPopulationResetForTest(); + ReferenceChainsTestAccessor::restartSearchForTest(); + reseedChunkLeakTags(topo); + seedLeakLiveness(leak_klass_id_out, topo); + // Poll once BEFORE the first pass of the new search: the candidate + // slot registers in the poll, and the pass's admission auto-mark + // requires the slot to exist (heapReferenceCallback's auto-mark + // guards on _candidate_count > 0). This is the pod's real ordering + // - the poll runs for minutes arming candidates before search #1. + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + } + + // Captures stdout per distinct ReferenceChainTracker/LivenessTracker + // line prefix (L4: the log-budget invariant - the round-15 800k/min + // flood and the round-18 tick stragglers become assertion failures). + struct StdoutCapture { + int saved_fd = -1; + int tmp_fd = -1; + char path[64] = {0}; + bool done = false; + StdoutCapture() { + snprintf(path, sizeof(path), "/tmp/podjar_stdout_XXXXXX"); + tmp_fd = mkstemp(path); + fflush(stdout); + saved_fd = dup(1); + dup2(tmp_fd, 1); + } + std::map perPrefixCounts() { + if (done) { + return {}; + } + done = true; + fflush(stdout); + dup2(saved_fd, 1); + close(saved_fd); + saved_fd = -1; + lseek(tmp_fd, 0, SEEK_SET); + std::map counts; + FILE *f = fdopen(tmp_fd, "r"); + char line[512]; + while (fgets(line, sizeof(line), f)) { + char cls[96], fn[96]; + if (sscanf(line, "[TEST::INFO] %95[^:]::%95[a-zA-Z]", cls, + fn) == 2) { + counts[std::string(cls) + "::" + fn]++; + } + } + fclose(f); + unlink(path); + tmp_fd = -1; + return counts; + } + }; +}; + +TEST_F(PodInAJarTest, SystemLivenessLeakChainsBuildAndCanaryResolves) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + + int cycles_run = 0; + while (cycles_run < 120 && + ReferenceChainsTestAccessor::resolvedChainCountForTest() < + (size_t)kChunks && + ReferenceChainsTestAccessor::searchStateForTest() == + SearchState::RUNNING) { + drivePodCycle(fake_now); + cycles_run++; + } + + // L1: every leak-tagged chunk's tag appears as a cached chain target + // (leak-correlated events - the pod's round-16 end-goal state). + auto targets = ReferenceChainsTestAccessor::resolvedChainTargetsForTest(); + for (int i = 0; i < kChunks; i++) { + jlong leak_tag = 1073741824LL + 100 + i; + EXPECT_NE(std::find(targets.begin(), targets.end(), (u64)leak_tag), + targets.end()) + << "leak tag " << leak_tag + << " never became a cached chain target after " << cycles_run + << " cycles"; + } + + // L2: the canary resolved for the LEAK klass (found bit set on its + // slot - the round-19 criterion; fails on any pre-788d7b2a7 build). + // Candidate slots persist across restarts BY DESIGN (see the slot + // registration comment: "a klass ... stops consuming pool tags even + // though its slot persists"), so the throwaway 999 slot from + // bringUpPod's phase 1 legitimately survives as an unfound slot. + int leak_slot = -1; + for (int s2 = 0; s2 < ReferenceChainsTestAccessor::candidateCountForTest(); + s2++) { + if (ReferenceChainsTestAccessor::candidateKlassIdForTest(s2) == + leak_klass_id) { + leak_slot = s2; + break; + } + } + ASSERT_GE(leak_slot, 0) << "leak klass never registered a candidate slot"; + EXPECT_TRUE(ReferenceChainsTestAccessor::candidateFoundBitsForTest() & + (1ULL << leak_slot)) + << "leak candidate slot never marked found despite leak-tag chains"; +} + +TEST_F(PodInAJarTest, SystemSearchCompletesNaturally) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + + int cycles = 0; + while (cycles < 200 && + ReferenceChainsTestAccessor::searchStateForTest() == + SearchState::RUNNING) { + drivePodCycle(fake_now); + cycles++; + } + + // L3: a healthy-topology search must end COMPLETED (all candidates + // found), never ABANDONED (TTL/frontier-cap). Pre-round-19 this line + // fails: the structurally unresolvable chase can only ever reach + // ABANDONED. + EXPECT_EQ((u8)SearchState::COMPLETED, + ReferenceChainsTestAccessor::searchStateForTest()) + << "search did not complete naturally within " << cycles + << " cycles (state=" + << (int)ReferenceChainsTestAccessor::searchStateForTest() << ")"; + EXPECT_GT(ReferenceChainsTestAccessor::passesRunForTest(), 0); +} + +TEST_F(PodInAJarTest, SystemRestartLeavesNothingBehind) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + for (int i = 0; i < 120 && + ReferenceChainsTestAccessor::resolvedChainCountForTest() == 0; + i++) { + drivePodCycle(fake_now); + } + ASSERT_GT(ReferenceChainsTestAccessor::resolvedChainCountForTest(), + (size_t)0); + + ReferenceChainsTestAccessor::restartSearchForTest(); + + // L6: the restart contract as a test instead of discipline. Resolved + // chains intentionally SURVIVE the restart (restartSearch's own + // comment: a chain describes a still-live sample and keeps being + // re-emitted across restarts) - the per-search state below must not. + EXPECT_GT(ReferenceChainsTestAccessor::resolvedChainCountForTest(), + (size_t)0) + << "resolved chains should persist across restarts by design"; + EXPECT_EQ((size_t)0, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_EQ((size_t)0, + ReferenceChainsTestAccessor::staticAnchorFreshQueueSizeForTest()); + EXPECT_TRUE(ReferenceChainsTestAccessor::anchorIndexIsEmptyForTest()); + for (int s = 0; s < 5; s++) { + EXPECT_EQ(0, + ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(s)) + << "discovered slot " << s << " survived restartSearch()"; + } +} + +TEST_F(PodInAJarTest, TopologyCapacityContractStaticAdmits) { + // The topology-builder capacity contract as a regression: the FULL + // buildLeakPod topology + one direct pass must admit the wrapper + // static (and with it the whole leak subtree). This is the test that + // caught the builder's original bug: classes capturing + // &node_tags[node] as their jclass identity dangle when addNode() + // grows the vector past capacity - the sweep then seeds no class, the + // gate closes on an empty lap, and NOTHING admits, silently (the + // state machine still reports a clean COMPLETED search). + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + EXPECT_GT(tags_ever_assigned[topo.wrapper_node], 0) + << "wrapper static never admitted (holder_cls_tag=" + << node_tags[topo.holder_class_node] + << " sweep_gate=" << ReferenceChainsTestAccessor::sweepGateStaticCountForTest() + << "/" << ReferenceChainsTestAccessor::sweepGateResolvedCountForTest() + << ")"; + + tracker->stop(); +} + +TEST_F(PodInAJarTest, SystemHealthyAppIsDormant) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // L8: the accidental 3h pod control, encoded - a heap with no leak + // candidates must produce ZERO searches (the candidate/generations + // gate holds the machinery dormant on a healthy app). + PodTopology topo = buildLeakPod(); + // No seedLeakLiveness(): no candidate ever qualifies. bringUpPod is + // also skipped - it seeds liveness; drive raw cycles instead. + (void)topo; + resetLivenessForPod(); + u64 fake_now = OS::nanotime(); + for (int i = 0; i < 20; i++) { + drivePodCycle(fake_now); + } + EXPECT_EQ(0, ReferenceChainsTestAccessor::passesRunForTest()) + << "searches ran on a healthy (candidate-less) app"; + EXPECT_EQ((size_t)0, + ReferenceChainsTestAccessor::resolvedChainCountForTest()); +} + +TEST_F(PodInAJarTest, SystemLogBudgetPerPass) { +#ifndef DEBUG + // The gtest binary compiles the main sources WITHOUT DEBUG (round-16 + // lesson: TEST_LOG is a no-op here) - the log-budget invariant can only + // run in a DEBUG-built test binary. Skip with the reason instead of + // asserting on captured output that structurally cannot exist. + GTEST_SKIP() << "log budget needs a DEBUG-built gtest binary"; +#else + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + PodTopology topo = buildLeakPod(); + u32 leak_klass_id; + u64 fake_now; + bringUpPod(topo, leak_klass_id, fake_now); + for (int i = 0; i < 3; i++) { + drivePodCycle(fake_now); + } + + // L4: capture one full cycle (pass + poll) and bound every distinct + // line shape. The round-15 flood was thousands of lines/pass of a + // single shape; the budget catches any per-object/per-entry line that + // escapes its tier. The harness runs at debug level 2 (pinned by the + // binary's static setenv), so every gated line is live here. + StdoutCapture capture; + drivePodCycle(fake_now); + auto counts = capture.perPrefixCounts(); + ASSERT_FALSE(counts.empty()) << "no diagnostics captured at level 2"; + for (const auto &kv : counts) { + EXPECT_LE(kv.second, 300) << "log line shape '" << kv.first + << "' fired " << kv.second + << " times in one pass+poll cycle"; + } +#endif +} + +class PollWatchedTargetsTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + JNINativeInterface_ jni_tbl{}; + JNIEnv_ mock_jni{}; + + std::unordered_map tags; + std::unordered_set dead_refs; // NewLocalRef returns NULL for these + + jvmtiEnv *orig_jvmti = nullptr; + static PollWatchedTargetsTest *active_fixture; + + void SetUp() override { + active_fixture = this; + ReferenceChainsTestAccessor::reset(); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(true); + + jvmti_tbl = jvmtiInterface_1_{}; + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.GetTag = &mock_GetTag; + jvmti_tbl.SetTag = &mock_SetTag; + jvmti_tbl.GetClassSignature = &mock_GetClassSignature; + jvmti_tbl.Deallocate = &mock_Deallocate; + mock_jvmti.functions = &jvmti_tbl; + orig_jvmti = VMTestAccessor::getJvmti(); + VMTestAccessor::setJvmti(&mock_jvmti); + + jni_tbl = JNINativeInterface_{}; + jni_tbl.NewLocalRef = &mock_NewLocalRef; + jni_tbl.DeleteLocalRef = &mock_DeleteLocalRef; + jni_tbl.GetObjectClass = &mock_GetObjectClass; + mock_jni.functions = &jni_tbl; + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + active_fixture = nullptr; + } + + static jvmtiError JNICALL mock_GetTag(jvmtiEnv *, jobject object, jlong *tag_ptr) { + auto it = active_fixture->tags.find(object); + *tag_ptr = it != active_fixture->tags.end() ? it->second : 0; + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_SetTag(jvmtiEnv *, jobject object, jlong tag) { + active_fixture->tags[object] = tag; + return JVMTI_ERROR_NONE; + } + + static jobject JNICALL mock_NewLocalRef(JNIEnv *, jobject ref) { + if (active_fixture->dead_refs.count(ref) > 0) { + return nullptr; + } + return ref; // identity passthrough - see this fixture's own comment + } + + static void JNICALL mock_DeleteLocalRef(JNIEnv *, jobject) { + // no-op: this fixture's fake jobject values are not real JNI refs. + } + + // pollWatchedTargets()'s diagnostic class-name lookup on the candidate's + // representative: this fixture's fake jobjects carry no real class + // identity, so a fixed non-null jclass plus a fixed signature is all + // GetObjectClass()/GetClassSignature() need to return for that lookup to + // complete without touching a real JVM. + static jclass JNICALL mock_GetObjectClass(JNIEnv *, jobject) { + return (jclass)0xC1A55; + } + + static jvmtiError JNICALL mock_GetClassSignature(jvmtiEnv *, jclass, + char **signature_ptr, + char **generic_ptr) { + *signature_ptr = strdup("Ltest/FakeKlass;"); + if (generic_ptr != nullptr) { + *generic_ptr = nullptr; + } + return JVMTI_ERROR_NONE; + } + + static jvmtiError JNICALL mock_Deallocate(jvmtiEnv *, unsigned char *mem) { + free(mem); + return JVMTI_ERROR_NONE; + } + + // Seeds LivenessTracker's real population table with a growing series + // for `klass_id` (20 strictly-increasing samples - satisfies + // selectLeakCandidates()'s min-fill, growth/floor magnitude, and + // sustained-trend hysteresis requirements, livenessTracker.h; 20 rather + // than the 10-sample minimum fill leaves comfortable margin past the + // hysteresis threshold rather than sitting exactly on its boundary) and + // points its representative at `rep`. + void seedGrowingCandidate(u32 klass_id, jweak rep) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + // Per-(klass, tid) qualification: selectLeakCandidates() also + // requires a qualifying allocating thread. These fixtures have + // no real tracked instances (mock JVMTI, no live heap), so a + // fixed synthetic tid exercises the gate without pretending to + // match any instance's real tid. + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest(nullptr, klass_id, rep); + } +}; + +PollWatchedTargetsTest *PollWatchedTargetsTest::active_fixture = nullptr; + +TEST_F(PollWatchedTargetsTest, EmitsEventForAlreadyDiscoveredCandidate) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + + // Model "already discovered by an ordinary runPass()": a root-level + // FrontierTable entry plus a matching GetTag() result, mirroring + // referenceChainJfrRoundtrip_ut.cpp's seeding style. + ASSERT_TRUE(tracker->frontierTable()->insert( + /*tag=*/7, /*parent_tag=*/0, /*referrer_klass=*/1, /*depth=*/0, + FrontierEntryState::EDGE)); + tags[obj] = 7; + // With class-tag matching, _candidate_frontier_tags must be set + // so buildCanaryChainEvent() can reconstruct the chain. + ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(7)); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, NoEventForNotYetDiscoveredCandidate) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + // GetTag() reports 0 (default) - no pass has reached this object yet. + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, NoDuplicateOnRepeatPoll) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; + ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + + // Klass 1 is still flagged (LivenessTracker's ranking doesn't know an + // event was already emitted for it) - a second, third, ... poll must + // not re-emit for the same target_tag. + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + EXPECT_EQ(7, ReferenceChainsTestAccessor::resolvedChainSourceTag(7)); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, SkipsCandidateWhoseWeakReferenceDied) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + dead_refs.insert(obj); // NewLocalRef(rep) -> NULL, as if GC'd + + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; // would resolve to a discovered tag, if it could resolve + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); + + tracker->stop(); +} + +// A cached chain must not re-emit forever: once the klass's representative +// stops resolving (collected, or LRU-evicted from LivenessTracker's +// population table - klassPopulationSetRepresentativeForTest()'s ref is the +// stand-in for either), the very next poll must prune it from +// _resolved_chains rather than leaving a dump keep re-emitting a chain for a +// sample that is gone (see _resolved_chains' own comment, referenceChains.h, +// and pollWatchedTargets()'s "candidate died, or was evicted" branch). +TEST_F(PollWatchedTargetsTest, ChainPersistsAfterRepresentativeDies) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; + ReferenceChainsTestAccessor::setCandidateFrontierTagForTest(0, 7); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + ASSERT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + + // The representative died. Per-instance caching: the chain persists + // (it describes a reference path that was valid at resolution time). + // It expires naturally when the search restarts and the frontier is + // wiped. The backend can filter stale chains via HeapLiveObject events. + dead_refs.insert(obj); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) + << "per-instance chains persist after representative dies; " + "they expire on search restart, not on representative death"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)); + + tracker->stop(); +} + +TEST_F(PollWatchedTargetsTest, NoOpWhenGcGenerationsDisabled) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Overrides this fixture's own SetUp() default - exercises the + // pollWatchedTargets() guard covering LivenessTracker's own + // _gc_generations gate (population tracking's own gate), not just this tracker's own _enabled. + LivenessTracker::instance()->setGcGenerationsForTest(false); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)obj); + ASSERT_TRUE(tracker->frontierTable()->insert( + 7, 0, 1, 0, FrontierEntryState::EDGE)); + tags[obj] = 7; + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::resolvedChainCount()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Resolved-chain cache (ReferenceChainTracker::cacheResolvedChain()/ +// drainPendingChainEvents(), referenceChains.cpp) - the mechanism that keeps a +// resolved chain alive across dumps so it re-emits into every JFR chunk the +// sample survives into, and keeps the dump-time writer's blocking +// lock-acquisition retry loop off the BFS scheduling thread (see +// _resolved_chains' own comment, referenceChains.h). These tests drive +// cacheResolvedChain()/drainPendingChainEvents() directly via +// ReferenceChainsTestAccessor rather than through the full +// pollWatchedTargets()/selectLeakCandidates() pipeline - the cache/overflow/ +// counter mechanism is independent of how a chain was produced, and driving it +// through hundreds of real LivenessTracker candidates just to reach +// MAX_RESOLVED_CHAINS would test the seeding helper, not this mechanism. +// --------------------------------------------------------------------------- + +class ResolvedChainCacheTest : public ::testing::Test { +protected: + void SetUp() override { + ReferenceChainsTestAccessor::reset(); + } + + void TearDown() override { + ReferenceChainsTestAccessor::reset(); + } + + static ReferenceChainEvent makeEvent(u64 target_tag) { + ReferenceChainEvent event; + event._target_tag = target_tag; + event._depth = 0; + return event; + } +}; + +// The defining property of the "stick around" model: a cached chain is +// re-emitted on every dump, not drained once. Two successive drains with no +// intervening change must BOTH return the cached chain, and the cache must +// stay populated afterwards (unlike the old queue, which emptied on drain). +TEST_F(ResolvedChainCacheTest, SnapshotReEmitsOnEveryDumpWithoutClearing) { + ReferenceChainsTestAccessor::cacheChain(/*source_tag=*/1, makeEvent(7), + /*search_ns=*/0); + + std::vector firstDump; + ReferenceChainsTestAccessor::drain(&firstDump); + ASSERT_EQ(1u, firstDump.size()); + EXPECT_EQ(7u, firstDump[0]._target_tag); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) + << "drain must not clear the cache"; + + // A second dump with nothing changed re-emits the same chain. + std::vector secondDump; + ReferenceChainsTestAccessor::drain(&secondDump); + ASSERT_EQ(1u, secondDump.size()); + EXPECT_EQ(7u, secondDump[0]._target_tag); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); +} + +// Re-resolving the same klass (a restart re-tags its sample, or a fresh walk +// finds a deeper path) refreshes its single cache slot in place rather than +// accumulating duplicates - so a dump re-emits one current chain per klass, +// not one per resolution. +TEST_F(ResolvedChainCacheTest, RefreshReplacesSameKlassInPlace) { + // Note on the collapsed signature: cacheResolvedChain() used to take a + // separate source_tag_val parameter, but every production caller passed + // the same value as the map key (the map is keyed by frontier tag - see + // _resolved_chains' own comment), so the pair was collapsed. The stored + // source_tag therefore always equals the key; the "rebuilt from a new + // tag" scenario is a DIFFERENT key after the restart (the old tag's + // entry expires via the search-generation stamp), not an in-place + // source_tag rewrite. + ReferenceChainsTestAccessor::cacheChain(1, makeEvent(7), 0); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_EQ(1, ReferenceChainsTestAccessor::resolvedChainSourceTag(1)); + + // Re-resolving the same key refreshes the event in place. + ReferenceChainsTestAccessor::cacheChain(1, makeEvent(9), 0); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::resolvedChainCount()) + << "refresh must overwrite, not append"; + EXPECT_EQ(1, ReferenceChainsTestAccessor::resolvedChainSourceTag(1)); + + std::vector dump; + ReferenceChainsTestAccessor::drain(&dump); + ASSERT_EQ(1u, dump.size()); + EXPECT_EQ(9u, dump[0]._target_tag); +} + +// Distinct klasses each get their own slot and all re-emit together in one +// dump (order is unspecified - the cache is a map keyed by klass_id). +TEST_F(ResolvedChainCacheTest, MultipleKlassesAllSnapshotTogether) { + ReferenceChainsTestAccessor::cacheChain(1, makeEvent(1), 0); + ReferenceChainsTestAccessor::cacheChain(2, makeEvent(2), 0); + ReferenceChainsTestAccessor::cacheChain(3, makeEvent(3), 0); + ASSERT_EQ(3u, ReferenceChainsTestAccessor::resolvedChainCount()); + + std::vector dump; + ReferenceChainsTestAccessor::drain(&dump); + ASSERT_EQ(3u, dump.size()); + std::set tags; + for (const auto &e : dump) { + tags.insert(e._target_tag); + } + EXPECT_EQ((std::set{1, 2, 3}), tags); +} + +// A brand-new klass arriving with the cache already at MAX_RESOLVED_CHAINS is +// dropped (and counted via REFERENCE_CHAIN_EVENTS_DROPPED, this codebase's own +// "dropped-event-without-counter" review lens) rather than evicting some other +// still-live sample's chain - but refreshing a klass that is already cached +// still succeeds even at capacity. +TEST_F(ResolvedChainCacheTest, OverflowDropsNewKlassButAllowsRefresh) { + const int cap = ReferenceChainsTestAccessor::maxResolvedChains(); + long long droppedBefore = Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED); + + for (int i = 0; i < cap; i++) { + ReferenceChainsTestAccessor::cacheChain((jlong)i, makeEvent((jlong)i), 0); + } + ASSERT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_EQ(droppedBefore, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)) + << "filling exactly to capacity must not drop anything yet"; + + // A brand-new klass at capacity is dropped and counted. + ReferenceChainsTestAccessor::cacheChain((jlong)cap, makeEvent((jlong)cap), 0); + EXPECT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()) + << "cache must stay capped, not grow past MAX_RESOLVED_CHAINS"; + EXPECT_EQ(droppedBefore + 1, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)); + EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag((u32)cap)); + + // Refreshing an already-cached entry at capacity must still succeed - it + // reuses that entry's existing slot rather than needing a free one. + // (Key 0 was pre-cached above; with the collapsed signature the stored + // source_tag always equals the key, so the assert checks identity, not + // the event's payload tag.) + ReferenceChainsTestAccessor::cacheChain(/*source_tag=*/0, makeEvent(999), 0); + EXPECT_EQ((size_t)cap, ReferenceChainsTestAccessor::resolvedChainCount()); + EXPECT_EQ(0, ReferenceChainsTestAccessor::resolvedChainSourceTag(0)); + std::vector refreshDump; + ReferenceChainsTestAccessor::drain(&refreshDump); + bool found999 = false; + for (const auto &e : refreshDump) { + if (e._target_tag == 999u) found999 = true; + } + EXPECT_TRUE(found999) << "the in-place refresh must update the event payload"; + EXPECT_EQ(droppedBefore + 1, Counters::getCounter(REFERENCE_CHAIN_EVENTS_DROPPED)) + << "an in-place refresh must not count as a drop"; +} + +// --------------------------------------------------------------------------- +// Pause-time pacing controller: pause-time-SLO feedback loop +// (ReferenceChainTracker::updatePacing(), referenceChains.cpp) - see that +// method's own comment (referenceChains.h) for the full mechanism. These +// tests drive updatePacing() directly with a synthetic sequence of "pass +// took Xms" wall-clock durations via ReferenceChainsTestAccessor (this +// file's existing pattern for reaching a private method/state - see the +// target-selection bridging step's hasResolvedChainForTag()/resolvedChainCount() above), +// reusing the ReferenceChainsTest fixture since updatePacing() itself makes +// no JVMTI calls. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsTest, PacingHoldsSteadyWhenPassesLandExactlyOnCeiling) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int startBudget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 startCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + ASSERT_EQ(4000, startBudget); // starts pinned at the configured ceiling + + // A pass landing exactly on the pause-time target is a zero error every + // call - the controller should never move away from its starting point, + // regardless of how many such passes are observed in a row. + for (int i = 0; i < 10; i++) { + ReferenceChainsTestAccessor::updatePacing(5 * 1000000ULL); // 5ms + EXPECT_EQ(startBudget, ReferenceChainsTestAccessor::effectiveBudget()); + EXPECT_EQ(startCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); + } + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, PacingShrinksBudgetAndWidensCadenceWhenOverCeiling) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int initialBudget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 initialCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + + // A pass taking 10x the pause-time ceiling, fed repeatedly (a constant + // input - the plan's own "does not oscillate indefinitely" scenario). + int lastBudget = initialBudget; + u64 lastCadence = initialCadence; + for (int i = 0; i < 20; i++) { + ReferenceChainsTestAccessor::updatePacing(50 * 1000000ULL); // 50ms + int budget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 cadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + EXPECT_LE(budget, lastBudget); // never grows while still over ceiling + EXPECT_GE(cadence, lastCadence); // never shrinks while still over ceiling + lastBudget = budget; + lastCadence = cadence; + } + + // Moved in the correct direction... + EXPECT_LT(lastBudget, initialBudget); + EXPECT_GT(lastCadence, initialCadence); + // ...and converged to a fixed point rather than oscillating: one more + // identical input produces no further change. + ReferenceChainsTestAccessor::updatePacing(50 * 1000000ULL); + EXPECT_EQ(lastBudget, ReferenceChainsTestAccessor::effectiveBudget()); + EXPECT_EQ(lastCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, PacingGrowsBudgetBackAndRelaxesCadenceWhenUnderCeiling) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=4000:pausetarget=5")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Start from a controlled below-ceiling/above-baseline point (as if an + // earlier over-ceiling run had already shrunk/widened them - see the + // previous test) with a freshly reset controller, rather than chaining + // directly off a constant-input sequence like the previous test's own: + // _pause_pid's integral state would otherwise still be recovering from + // that sequence's windup for many iterations after switching to a + // smaller-magnitude error, muddying this test's per-step "moves in the + // correct direction every step" assertions with a transient this test + // is not about. + ReferenceChainsTestAccessor::setEffectiveBudget(2400); + ReferenceChainsTestAccessor::setEffectiveCadenceNs( + 2 * ReferenceChainsTestAccessor::baselineCadenceNs()); + ReferenceChainsTestAccessor::resetPacingController(); + int shrunkBudget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 widenedCadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + + // Now feed passes comfortably under the ceiling, repeatedly (a constant + // input, to check convergence rather than oscillation). 200 iterations - + // more than PacingShrinksBudgetAndWidensCadenceWhenOverCeiling needs - + // because this scenario's error magnitude (pausetarget=5 vs. an + // effectively-instant 0ms pass) is smaller, so the cadence side takes + // more iterations to fully unwind down to MIN_EFFECTIVE_CADENCE_NS, and + // because the borrowed-budget distance to close (configured budget * + // (BORROW_CEILING_MULTIPLIER - 1)) scales with the configured budget + // while the PID's per-pass step size does not. + int lastBudget = shrunkBudget; + u64 lastCadence = widenedCadence; + for (int i = 0; i < 200; i++) { + ReferenceChainsTestAccessor::updatePacing(0); // effectively instant + int budget = ReferenceChainsTestAccessor::effectiveBudget(); + u64 cadence = ReferenceChainsTestAccessor::effectiveCadenceNs(); + EXPECT_GE(budget, lastBudget); // never shrinks while comfortably under + EXPECT_LE(cadence, lastCadence); // never widens while comfortably under + lastBudget = budget; + lastCadence = cadence; + } + + // Moved in the correct direction... and, since 50 identical + // comfortably-under-target passes is well past BORROW_WARMUP_PASSES, + // past the configured ceiling too - budget-borrowing lets it converge at + // the borrowed ceiling (configured budget * multiplier) instead of + // stalling at the plain configured budget. + EXPECT_GT(lastBudget, shrunkBudget); + EXPECT_EQ(4000 * ReferenceChainsTestAccessor::borrowCeilingMultiplier(), lastBudget); + EXPECT_LT(lastCadence, widenedCadence); + // ...and converged: one more identical input produces no further change. + ReferenceChainsTestAccessor::updatePacing(0); + EXPECT_EQ(lastBudget, ReferenceChainsTestAccessor::effectiveBudget()); + EXPECT_EQ(lastCadence, ReferenceChainsTestAccessor::effectiveCadenceNs()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, MaybeRevokeBorrowForRootEnumPassPreservesBorrowAtBoundary) { + Arguments args; + // BORROW_UNDER_TARGET_FRACTION (referenceChains.h) is 0.5, so with + // pausetarget=10 the comfortably-under-target boundary is exactly 5ms. + ASSERT_FALSE(args.parse("referencechains=true:budget=1000:pausetarget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ReferenceChainsTestAccessor::setBorrowedBudget(500); + ReferenceChainsTestAccessor::setConsecutiveUnderTargetPasses(5); + + // Exactly at the boundary: comfortably_under_target's `<=` check must + // still treat this as comfortably under, so the borrow is preserved. + ReferenceChainsTestAccessor::maybeRevokeBorrowForRootEnumPass(5 * 1000000ULL); + EXPECT_EQ(500, ReferenceChainsTestAccessor::borrowedBudget()); + EXPECT_EQ(5, ReferenceChainsTestAccessor::consecutiveUnderTargetPasses()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsTest, MaybeRevokeBorrowForRootEnumPassRevokesJustPastBoundary) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:budget=1000:pausetarget=10")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ReferenceChainsTestAccessor::setBorrowedBudget(500); + ReferenceChainsTestAccessor::setConsecutiveUnderTargetPasses(5); + ReferenceChainsTestAccessor::setEffectiveBudget(1500); // as if borrow had raised the ceiling + + // Just past the boundary: no longer comfortably under target, so the + // grant is revoked immediately, including re-clamping _effective_budget + // down to the plain (non-borrowed) budget rather than leaving it + // borrow-inflated until the next ordinary pass's updatePacing() call. + ReferenceChainsTestAccessor::maybeRevokeBorrowForRootEnumPass(6 * 1000000ULL); + EXPECT_EQ(0, ReferenceChainsTestAccessor::borrowedBudget()); + EXPECT_EQ(0, ReferenceChainsTestAccessor::consecutiveUnderTargetPasses()); + EXPECT_EQ(1000, ReferenceChainsTestAccessor::effectiveBudget()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// PainBudget (painBudget.h) - standalone, no ReferenceChainTracker singleton +// involved. A leaky bucket over cost (ms), not an event rate: spend() +// records how much an operation cost, canStartNow() drains the balance by +// elapsed wall-clock time at the configured refill rate and reports whether +// the debt has cleared. +// --------------------------------------------------------------------------- + +TEST(PainBudgetTest, ClearBeforeAnythingIsEverSpent) { + PainBudget budget(0.01); + EXPECT_TRUE(budget.canStartNow(1000)); +} + +TEST(PainBudgetTest, SpendCreatesDebtThatBlocksAnImmediateSecondCall) { + PainBudget budget(0.01); // 1% + ASSERT_TRUE(budget.canStartNow(1000)); // establishes the drain baseline + budget.spend(100); // 100ms of debt + // No time has elapsed since the baseline call above - the debt cannot + // have drained at all yet. + EXPECT_FALSE(budget.canStartNow(1000)); +} + +TEST(PainBudgetTest, DebtDrainsProportionallyToElapsedTimeAndRefillRate) { + PainBudget budget(0.01); // 1% -> 1ms of debt needs 100ms elapsed to clear + ASSERT_TRUE(budget.canStartNow(0)); + budget.spend(10); // 10ms of debt -> needs 1000ms elapsed to fully clear + EXPECT_FALSE(budget.canStartNow(500ULL * 1000000ULL)); // 500ms elapsed - not enough + EXPECT_TRUE(budget.canStartNow(1500ULL * 1000000ULL)); // 1500ms total - enough +} + +TEST(PainBudgetTest, ZeroRefillRateNeverClearsDebt) { + PainBudget budget(0.0); + ASSERT_TRUE(budget.canStartNow(0)); + budget.spend(1); + // An enormous elapsed time still drains nothing at a 0 refill rate. + EXPECT_FALSE(budget.canStartNow(1000000000000ULL)); +} + +// --------------------------------------------------------------------------- +// PriorityExpandSet (referenceChains.h) - the fixed-capacity open-addressed +// set behind _priority_expand. Reviewed gaps: no test referenced the class +// at all, so the probe/insert/clear/rebuildFrom paths and the full-table +// self-termination were only covered indirectly. Driven through the test +// accessor (the class is a private member of the tracker singleton); all +// operations here are pure in-process logic, no JVMTI. +// --------------------------------------------------------------------------- +TEST(PriorityExpandSetTest, InsertContainsClearRoundTrip) { + ReferenceChainsTestAccessor::pesClear(); + + EXPECT_FALSE(ReferenceChainsTestAccessor::pesContains(42)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesInsert(42)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(42)); + // Idempotent insert: already present returns false. + EXPECT_FALSE(ReferenceChainsTestAccessor::pesInsert(42)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesInsert(43)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(43)); + + ReferenceChainsTestAccessor::pesClear(); + EXPECT_FALSE(ReferenceChainsTestAccessor::pesContains(42)) + << "clear must empty the set"; + EXPECT_FALSE(ReferenceChainsTestAccessor::pesContains(43)); + + // Re-insert after clear (slot reuse must work). + EXPECT_TRUE(ReferenceChainsTestAccessor::pesInsert(42)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(42)); + ReferenceChainsTestAccessor::pesClear(); +} + +TEST(PriorityExpandSetTest, RebuildFromMatchesQueueContents) { + ReferenceChainsTestAccessor::pesClear(); + + std::deque queue = {10, 20, 30}; + ReferenceChainsTestAccessor::pesRebuildFrom(queue); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(10)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(20)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(30)); + EXPECT_FALSE(ReferenceChainsTestAccessor::pesContains(40)); + + // Pop the front (rotation collection's ordinary shape) and rebuild: + // membership must track the queue exactly - 10 is gone. + queue.pop_front(); + ReferenceChainsTestAccessor::pesRebuildFrom(queue); + EXPECT_FALSE(ReferenceChainsTestAccessor::pesContains(10)) + << "stale member after rebuildFrom"; + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(20)); + EXPECT_TRUE(ReferenceChainsTestAccessor::pesContains(30)); + ReferenceChainsTestAccessor::pesClear(); +} + +// A completely full table must not livelock the linear-probe loop: insert() +// is required to be self-terminating (returning false - the same "treated +// as not-queued" degradation a full deque push already accepts) even if a +// call site breaks the external "<= PRIORITY_EXPAND_CAP by construction" +// invariant. Fills to capacity with DISTINCT tags, then asserts one more +// insert on a fresh tag terminates (and returns false), and that a +// duplicate of an existing member still returns false via the key match. +// ClassTagAllocator (classTagAllocator.h) - the process-wide negative +// class-object tag allocator shared by ReferenceChainTracker and +// LivenessTracker. Reviewed gap: no test exercised its conventions +// directly. resetForTest() makes each test's minting sequence +// deterministic despite the process-wide counter. +TEST(ClassTagAllocatorTest, MintsStrictlyNegativeStrictlyDecreasingTags) { + ClassTagAllocator::resetForTest(); + jlong first = ClassTagAllocator::next(); + jlong second = ClassTagAllocator::next(); + EXPECT_LT(first, 0) << "class tags must be negative (heapReferenceCallback" + "() dispatches on the sign)"; + EXPECT_LT(second, first) << "tags must decrease monotonically (no reuse)"; + ClassTagAllocator::resetForTest(); + EXPECT_EQ(first, ClassTagAllocator::next()) + << "resetForTest must restart the sequence at the same value"; +} + +TEST(PriorityExpandSetTest, InsertOnFullTableTerminatesWithoutLivelock) { + ReferenceChainsTestAccessor::pesClear(); + + const int capacity = 1 << 11; // 2048 slots (SLOT_SHIFT == 11) + for (int i = 0; i < capacity; i++) { + // Distinct tags spread by Fibonacci hashing - sequential ints are + // the worst case for proving full-table behavior is probe-safe, so + // use them deliberately: they land on distinct slots only via the + // mix(), exercising the wrap path. + ASSERT_TRUE(ReferenceChainsTestAccessor::pesInsert(1000 + (jlong)i * 7919)) + << "insert must succeed until the table is full (i=" << i << ")"; + } + // Table full: a fresh tag must return false promptly (no hang). + EXPECT_FALSE(ReferenceChainsTestAccessor::pesInsert(987654321)); + // A duplicate of an existing member still returns false via the key + // match (found while probing). + EXPECT_FALSE(ReferenceChainsTestAccessor::pesInsert(1000)); + ReferenceChainsTestAccessor::pesClear(); + EXPECT_FALSE(ReferenceChainsTestAccessor::pesContains(1000)); +} + +// --------------------------------------------------------------------------- +// Search restart (referenceChains.h's own header comment: gating a +// restarted search's first pass on LivenessTracker already reporting a leak +// candidate, plus the PainBudget cooldown above, so a search that already +// walked the whole reachable graph once does not do so again indefinitely +// without a reason). Uses an "empty reachable graph" FollowReferences mock +// (no callback invocations at all) to reach SearchState::COMPLETED in one +// call - the simplest way to drive a search to a terminal state without +// ReferenceChainsBfsTest's full scripted-graph machinery, which exists for +// chain-reconstruction coverage this suite does not need. +// --------------------------------------------------------------------------- + +class SearchRestartTest : public ::testing::Test { +protected: + jvmtiInterface_1_ jvmti_tbl{}; + _jvmtiEnv mock_jvmti{}; + jvmtiEnv *orig_jvmti = nullptr; + + void SetUp() override { + ReferenceChainsTestAccessor::reset(); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + + jvmti_tbl = jvmtiInterface_1_{}; + jvmti_tbl.SetEventNotificationMode = &mock_SetEventNotificationMode; + jvmti_tbl.GetLoadedClasses = &mock_GetLoadedClasses; + jvmti_tbl.FollowReferences = &mock_FollowReferences; + jvmti_tbl.IterateOverReachableObjects = &mock_IterateOverReachableObjects; + jvmti_tbl.GetAvailableProcessors = &mock_GetAvailableProcessors; + mock_jvmti.functions = &jvmti_tbl; + orig_jvmti = VMTestAccessor::getJvmti(); + VMTestAccessor::setJvmti(&mock_jvmti); + } + + void TearDown() override { + VMTestAccessor::setJvmti(orig_jvmti); + LivenessTracker::instance()->klassPopulationResetForTest(); + LivenessTracker::instance()->setGcGenerationsForTest(false); + // UrgentOOMProjectionBypassesCandidateGate below sets this - reset it + // here (TearDown always runs, even after a fatal ASSERT_* return) + // rather than as a trailing statement in that test body, so a failed + // assertion can't leak a stale max-heap value into the next test + // sharing this singleton. + LivenessTracker::instance()->setMaxHeapBytesForTest(-1); + } + + // No loaded classes to resolve - resolveLoadedClasses() reports 0 and + // does nothing further. + static jvmtiError JNICALL mock_GetLoadedClasses(jvmtiEnv *, jint *count, + jclass **out) { + *count = 0; + *out = nullptr; + return JVMTI_ERROR_NONE; + } + + // ReferenceChainTracker::start() -> autoTuneDefaults() queries this + // whenever LivenessTracker reports a max heap > 0 - which + // UrgentOOMProjectionBypassesCandidateGate below sets. This fixture's + // table predates that query (facdc70c0, 2026-08-17): the unwired entry + // null-crashed that test at start() - before its subject ever ran - + // for every full-suite run since. A fixed single processor keeps the + // auto-tuned pause/pain values deterministic regardless of the host. + static jvmtiError JNICALL mock_GetAvailableProcessors(jvmtiEnv *, + jint *nprocs) { + *nprocs = 1; + return JVMTI_ERROR_NONE; + } + + // Never invokes the callback - models a heap with nothing reachable from + // any root, so the very first pass completes immediately (0 admitted + // edges, not truncated). + static jvmtiError JNICALL mock_FollowReferences( + jvmtiEnv *, jint, jclass, jobject, const jvmtiHeapCallbacks *, + const void *) { + return JVMTI_ERROR_NONE; + } + + // runPassManualWalk()'s root enumeration - never invokes the root + // callback, same "nothing reachable from any root" heap model as + // mock_FollowReferences() above, so the first pass still completes + // immediately with 0 admitted edges. + static jvmtiError JNICALL mock_IterateOverReachableObjects( + jvmtiEnv *, jvmtiHeapRootCallback, jvmtiStackReferenceCallback, + jvmtiObjectReferenceCallback, const void *) { + return JVMTI_ERROR_NONE; + } + + // Same seeding helper as PollWatchedTargetsTest above (20 strictly- + // increasing samples - satisfies selectLeakCandidates()'s min-fill, + // growth/floor magnitude, and sustained-trend hysteresis requirements). + void seedGrowingCandidate(u32 klass_id, jweak rep) { + int slot; + bool created; + for (u16 i = 1; i <= 20; i++) { + LivenessTracker::instance()->klassPopulationRecordForTest( + klass_id, i, i, &slot, &created); + // Per-(klass, tid) qualification: selectLeakCandidates() also + // requires a qualifying allocating thread. These fixtures have + // no real tracked instances (mock JVMTI, no live heap), so a + // fixed synthetic tid exercises the gate without pretending to + // match any instance's real tid. + LivenessTracker::instance()->tidTrendRecordForTest( + klass_id, /*tid=*/4242, (u32)i, (u64)i); + } + LivenessTracker::instance()->klassPopulationSetRepresentativeForTest(nullptr, klass_id, rep); + } +}; + +TEST_F(SearchRestartTest, WithoutGenerationsSignalRestartStaysUnconditional) { + // gc_generations off (this fixture's SetUp default): canAffordNewSearch() + // has no candidate signal to gate on at all, so a terminal search is + // immediately eligible to restart - preserves this tracker's pre-restart + // behavior for a referencechains-without-generations setup (this class's + // own header comment, last paragraph). + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, GenerationsEnabledButNoCandidateBlocksFirstSearch) { + // A brand-new tracker must not pay for the initial whole-heap + // walk/tagging pass either when there is no leak candidate yet - + // shouldRunPass()'s !_search_started branch now shares + // canAffordNewSearch() with the restart gate below (this class's own + // header comment). + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + EXPECT_EQ(0, tracker->passesRun()); + + int fake_object_storage = 0; + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(2)); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, GenerationsEnabledButNoCandidateBlocksRestart) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // No leak candidate flagged - nothing to justify the cost of a restart. + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, RestartsOnceACandidateAppearsAndResetsPerSearchState) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + ASSERT_EQ(1, tracker->passesRun()); + + int fake_object_storage = 0; + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); // restartSearch() runs inline + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + EXPECT_EQ(0, tracker->passesRun()); // restartSearch() zeroed per-search state + + // The next runPass() call takes the "first pass of a search" branch + // again, exactly like a brand-new tracker. + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + EXPECT_EQ(1, tracker->passesRun()); + + tracker->stop(); +} + +TEST_F(SearchRestartTest, PainBudgetBlocksARestartUntilItDrains) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:painbudget=1")); // 1% + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + seedGrowingCandidate(/*source_tag=*/1, /*rep=*/(jweak)&fake_object_storage); + + // First-ever search: called via runPass() directly here, bypassing + // shouldRunPass()'s canAffordNewSearch() gate entirely - the candidate + // seeded above would satisfy that gate anyway (this class's own header + // comment). + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Restart #1: _safepoint_pain_budget has never had anything spent into it yet, so + // this is always immediately affordable regardless of this first + // search's own cost - the cost a search incurs only debits the *next* + // restart's affordability (restartSearch()'s own spend-then-reset + // order), not its own. + ASSERT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Pretend this second search cost 1000ms of safepoint time - a mocked + // FollowReferences call in this fixture takes ~0 real wall-clock time, + // so this accessor stands in for what a real, expensive pass would have + // accumulated into _search_pain_ms on its own. + ReferenceChainsTestAccessor::setSearchPainMs(1000); + + // Restart #2: the terminal gate charges the finished search's OWN 1000ms + // cost BEFORE checking affordability (canAffordNewSearch() must see the + // cost of the search that just ended, or an expensive search would earn + // one free immediate successor). At 1% refill, 1000ms of debt needs + // 100000ms (1e11ns) of elapsed wall-clock time to clear - 1ns later is + // nowhere close. + EXPECT_FALSE(ReferenceChainsTestAccessor::shouldRunPass(2)); + EXPECT_EQ(SearchState::COMPLETED, tracker->searchState()); + + // Well past the drain point - the debt has cleared, restart #2 proceeds. + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1ULL + 200000000000ULL)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, nullptr)); + ASSERT_EQ(SearchState::COMPLETED, tracker->searchState()); + + tracker->stop(); +} + +// hasLeakSignal()'s OOM_URGENT_THRESHOLD_S fast path (referenceChains.h/.cpp): +// a heap-wide leak growing fast enough to project exhaustion sooner than the +// threshold must start a search immediately, without waiting for any klass +// to clear selectLeakCandidates()'s own per-klass ring-fill/hysteresis gate - +// this is the aggressive-leak gap GenerationsEnabledButNoCandidateBlocksFirstSearch +// above documents for the non-urgent case. Deliberately seeds no candidate at +// all (LivenessTracker::instance()->klassPopulationResetForTest() in SetUp +// leaves the population table empty) so this test can only pass via the +// heap-floor projection, never via selectLeakCandidates() falling back to a +// real candidate. +TEST_F(SearchRestartTest, UrgentOOMProjectionBypassesCandidateGate) { + LivenessTracker::instance()->setGcGenerationsForTest(true); + constexpr u64 SEC_NS = 1000000000ULL; + constexpr u64 MiB = 1ULL << 20; + // Same worked example as livenessTracker_ut.cpp's + // SecondsToOOMTest.RisingFloorProjectsExpectedSeconds: 700MiB rise over + // 7s against a 2800MiB max heap projects to 10s - comfortably under + // OOM_URGENT_THRESHOLD_S (5 minutes). + LivenessTracker::instance()->setMaxHeapBytesForTest((jlong)(2800 * MiB)); + for (int i = 0; i < 10; i++) { + LivenessTracker::instance()->heapFloorRecordForTest( + 1000 * MiB + (u64)i * 100 * MiB, (u64)i * SEC_NS); + } + + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + EXPECT_TRUE(ReferenceChainsTestAccessor::shouldRunPass(1)); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Durability re-verification (correctness hardening). +// +// These tests drive maybeUpgradeRootAttachedRootKind()/ +// collectStaleRootKindEntriesForRotation() directly via +// ReferenceChainsTestAccessor rather than through a full +// IterateOverReachableObjects()-driven runPassManualWalk() pass: neither +// IterateOverReachableObjects nor FollowReferences-as-a-safepoint-pin is +// mocked in this file (see the fixture's own FollowReferences-only mock +// rationale above), and both methods are pure FrontierTable/queue logic with +// no JVMTI dependency of their own - the same rationale +// admitObject()/rootKindDurability() being free of any callback shape +// already established for this subsystem. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, StaleRootAttributionUpgradesOnRediscovery) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Synthetic stack-local root: admitted, root-attached (parent_tag == 0), + // its owning frame has since "gone away" from the design doc's scenario + // (nothing further to model here - the entry simply stays as-is until a + // more durable root is discovered). + jlong tag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, /*parent_tag=*/0, /*depth=*/0, + FrontierEntryState::EXPANDED, JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + // A second, equally-or-less durable root discovery does not overwrite + // the recorded root_kind. + EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, tag, JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(tag, &entry)); + EXPECT_EQ(JVMTI_HEAP_REFERENCE_STACK_LOCAL, entry.root_kind); + + // A durable root (JNI global) attaching to the same object upgrades it. + EXPECT_TRUE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, tag, JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + ASSERT_TRUE(frontier->lookup(tag, &entry)); + EXPECT_EQ(JVMTI_HEAP_REFERENCE_JNI_GLOBAL, entry.root_kind); + EXPECT_EQ(0, entry.parent_tag); // still root-attached, unchanged + + // An even less durable root discovered afterwards cannot downgrade it. + EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, tag, JVMTI_HEAP_REFERENCE_MONITOR)); + ASSERT_TRUE(frontier->lookup(tag, &entry)); + EXPECT_EQ(JVMTI_HEAP_REFERENCE_JNI_GLOBAL, entry.root_kind); + + tracker->stop(); +} + +// Exercises the invariant conflict that durability re-verification exists to +// catch: a non-root +// entry (parent_tag != 0) rediscovered as if via a root context must never +// have its root_kind overwritten - doing so would leave a non-zero root_kind +// on an entry nothing else treats as root-attached (referenceChains.h's +// FrontierEntry::root_kind comment), since this mutator never touches +// parent_tag. This is the edge-based, non-root-Y re-expansion case the +// option (a) resolution above exists for. +TEST_F(ReferenceChainsBfsTest, NonRootAttachedEntryNeverUpgraded) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Parent Y (root-attached) and child X, admitted the way frontier + // re-expansion admits a non-root child: non-root + // (parent_tag == Y's tag), root_kind == 0. + jlong yTag = 1; + jlong xTag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, yTag, /*parent_tag=*/0, /*depth=*/0, + FrontierEntryState::EXPANDED, JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, xTag, /*parent_tag=*/yTag, /*depth=*/1, + FrontierEntryState::EXPANDED, /*root_kind=*/0)); + + // Re-expanding Y rediscovers an edge to X (already tracked) - even if + // this rediscovery is (incorrectly) attempted with a durable root_kind, + // it must be rejected because X is not root-attached. + EXPECT_FALSE(ReferenceChainsTestAccessor::maybeUpgradeRootAttachedRootKind( + frontier, xTag, JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(xTag, &entry)); + EXPECT_EQ(0, entry.root_kind); + EXPECT_EQ(yTag, entry.parent_tag); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, RotationSelectsOnlyTransientExpandedRootAttachedEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Eligible: root-attached, EXPANDED, transient root_kind. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + // Not eligible: durable root_kind. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + // Not eligible: transient but still FRONTIER, not yet EXPANDED. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 3, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + // Not eligible: transient root_kind but not root-attached. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 4, /*parent_tag=*/1, 1, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + // Eligible: root-attached, EXPANDED, transient (JNI local this time). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + + std::vector selected = + ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation(10); + std::sort(selected.begin(), selected.end()); + EXPECT_EQ((std::vector{1, 5}), selected); + + // Selected tags are queued for re-expansion, exactly like an ordinary + // admission would queue a newly-discovered tag. + EXPECT_EQ(2u, ReferenceChainsTestAccessor::priorityExpandSize()); + + tracker->stop(); +} + +// N transient-root_kind entries, rotation size R: every entry must be +// selected at least once within ceil(N/R) calls, regardless of where the +// cursor happened to start. +TEST_F(ReferenceChainsBfsTest, RotationCoversAllEntriesWithinCeilNOverR) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + const int N = 10; + const int R = 3; + for (jlong tag = 1; tag <= N; tag++) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + } + + std::unordered_set covered; + int calls = (N + R - 1) / R; + for (int i = 0; i < calls; i++) { + std::vector selected = + ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation(R); + for (jlong tag : selected) { + covered.insert(tag); + } + } + EXPECT_EQ((size_t)N, covered.size()); + + tracker->stop(); +} + +// collectStaleExpandedEntriesForRotation()'s own EXPANDED-only criterion is a +// strict superset of collectStaleRootKindEntriesForRotation()'s (which also +// requires parent_tag == 0 and a transient root_kind), and runPassManualWalk() +// calls the root-kind collector first, into the very same _priority_expand +// deque. Without a dedup check, a tag the root-kind collector already queued +// would be queued a second time by the EXPANDED-only sweep, wasting one of +// expandFrontier()'s per-entry batch slots on an already-EXPANDED tag every +// pass. This drives both collectors back-to-back, the way runPassManualWalk() +// does, and asserts _priority_expand ends up with no duplicate tags. +TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationDoesNotDuplicateRootKindSelection) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Eligible for both collectors: EXPANDED, root-attached, transient + // root_kind - exactly the overlap collectStaleRootKindEntriesForRotation() + // will pick up first. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + // Eligible only for the EXPANDED-only sweep: EXPANDED but not root-attached. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, /*parent_tag=*/1, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0)); + + std::vector root_kind_selected = + ReferenceChainsTestAccessor::collectStaleRootKindEntriesForRotation( + ReferenceChainsTestAccessor::rootKindRotationBudget()); + EXPECT_EQ((std::vector{1}), root_kind_selected); + + std::vector stale_expanded_selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( + ReferenceChainsTestAccessor::staleExpandedRotationBudget()); + // Tag 1 is already queued from the root-kind collector above and must not + // be selected again; tag 2 is newly discovered by this sweep. + EXPECT_EQ((std::vector{2}), stale_expanded_selected); + + std::vector queued = ReferenceChainsTestAccessor::priorityExpandContents(); + EXPECT_EQ((std::vector{1, 2}), queued); + std::unordered_set unique_queued(queued.begin(), queued.end()); + EXPECT_EQ(queued.size(), unique_queued.size()); + + tracker->stop(); +} + +// A tag left over in _priority_expand from a prior pass's truncated +// expandFrontier() batch (see expandFrontier()'s own "leave the batch at the +// front of the source queue for a later pass to retry" comment) must also be +// skipped by collectStaleExpandedEntriesForRotation() - not just tags queued +// by collectStaleRootKindEntriesForRotation() earlier in the same call. +TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationSkipsPreexistingQueueEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + // Simulate a truncated batch from a prior pass still sitting at the front + // of _priority_expand, without going through the root-kind collector at + // all - the leftover entry alone must still be enough to suppress a + // duplicate. + ReferenceChainsTestAccessor::pushPriorityExpand(1); + + std::vector stale_expanded_selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( + ReferenceChainsTestAccessor::staleExpandedRotationBudget()); + EXPECT_TRUE(stale_expanded_selected.empty()); + + std::vector queued = ReferenceChainsTestAccessor::priorityExpandContents(); + EXPECT_EQ((std::vector{1}), queued); + + tracker->stop(); +} + +// End-to-end proof of the prof-analyzer-hotdog-jb pod's actual leak shape: +// a static-field-rooted collection (like ProfileAnalyzer.LEAK_BUFFER) whose +// owning node is admitted and fully EXPANDED once, then has a *new* element +// appended to it afterward - mirroring a Java List field being mutated in +// place, never reassigned, well after admitStaticFieldRoots()'s one-time +// sweep. The critical property under test is that this new element is +// discovered by collectStaleExpandedEntriesForRotation()'s rotation without +// the overall search ever reaching SearchState::COMPLETED - i.e. without +// requiring a full heap walk to finish, which on a multi-GiB heap can take +// far longer than the pod can tolerate between the leaked field's own +// growth events. A large distractor root chain (never fully drained within +// this test's bounded pass loops) keeps the search perpetually RUNNING so +// this property is exercised directly, not sidestepped by letting the +// search finish and then trivially re-discovering everything from scratch. +TEST_F(ReferenceChainsBfsTest, RotationDiscoversLateElementOfExpandedStaticFieldCollectionWithoutSearchCompleting) { + Arguments args; + // budget=8 -> rotation_reserved_budget = min(8/2, 272) = 4, ordinary = 4: + // both slices non-zero, unlike a budget=1 pattern which would zero out + // rotation's reserved slice entirely (min(0, 272) == 0). + ASSERT_FALSE(args.parse("referencechains=true:hops=5000:budget=8:firstpassbudget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + int listNode = addNode(); + int seedChildNode = addNode(); + int lateChildNode = addNode(); + + // Distractor chain: a long, independently root-seeded chain that never + // fully drains within this test's bounded pass loops below, so the + // overall search always has forward progress available and never + // reaches SearchState::COMPLETED (nor NO_PROGRESS_PASS_LIMIT-triggered + // ABANDONED) purely as a side effect of this test's own loop bounds. + const int kDistractorNodes = 500; + std::vector distractor(kDistractorNodes); + for (int i = 0; i < kDistractorNodes; i++) { + distractor[i] = addNode(); + } + + // addClass() captures classNode's address in node_tags' backing storage - + // must come after every addNode() call above (including the distractor + // loop), or a later push_back reallocating node_tags would silently + // leave this pointer dangling (indexOfNode() would then never match it). + addClass((void *)&node_tags[classNode], "Lcom/rc/statics/GrowingListHolder;"); + + script = { + // listNode is retained only via classNode's static field - the same + // shape as DiscoversObjectRetainedOnlyByStaticField above. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, listNode, -1}, + // listNode's one pre-existing element, discovered the first time + // listNode itself is expanded. + {JVMTI_HEAP_REFERENCE_FIELD, listNode, seedChildNode, -1}, + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, distractor[0], -1}, + }; + for (int i = 0; i + 1 < kDistractorNodes; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, distractor[i], distractor[i + 1], -1}); + } + + // Phase 1: run passes until listNode has been fully expanded (its one + // pre-existing child discovered), without ever letting the search + // complete. + bool truncated = true; + FrontierEntry listEntry{}; + bool listExpanded = false; + for (int i = 0; i < 200 && !listExpanded; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + jlong listTag = tags_ever_assigned[listNode]; + if (listTag != 0 && tracker->frontierTable()->lookup(listTag, &listEntry) + && listEntry.state == FrontierEntryState::EXPANDED) { + listExpanded = true; + } + } + ASSERT_TRUE(listExpanded); + ASSERT_NE(0, tags_ever_assigned[seedChildNode]); + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + + // Phase 2: simulate a new element appended to the leaking static + // field's list *after* listNode's one-time expansion - the exact + // "growing collection" shape found in the real pod's leak generator. + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, listNode, lateChildNode, -1}); + + for (int i = 0; i < 200 && tags_ever_assigned[lateChildNode] == 0; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + } + + // The late element was discovered purely via rotation re-expanding + // listNode - and, critically, without the search ever completing (no + // dependency on a full heap walk finishing). + ASSERT_NE(0, tags_ever_assigned[lateChildNode]); + EXPECT_EQ(SearchState::RUNNING, tracker->searchState()); + + std::vector chain; + ASSERT_TRUE(tracker->frontierTable()->reconstructChain( + tags_ever_assigned[lateChildNode], &chain)); + FrontierEntry lateEntry{}; + ASSERT_TRUE(tracker->frontierTable()->lookup( + tags_ever_assigned[lateChildNode], &lateEntry)); + EXPECT_EQ(tags_ever_assigned[listNode], lateEntry.parent_tag); + + tracker->stop(); +} + +// Proof of the fix for the actual prof-analyzer-hotdog-jb stall: +// collectStaleExpandedEntriesForRotation() (referenceChains.cpp) used to +// always rescan FrontierTable slots starting from tag 1, unlike its sibling +// collectStaleRootKindEntriesForRotation() which already carried its own +// persistent cursor. isQueuedForRotation()'s dedup check only looks at the +// CURRENT pass's _priority_expand (drained by expandFrontier() at the end of +// that same pass - see clearPriorityExpand()'s own comment), so without a +// cursor, later passes had no memory of what earlier passes already +// selected: whenever a real heap's frontier table held +// STALE_EXPANDED_ROTATION_BUDGET (256) or more low-tag EXPANDED entries that +// stay EXPANDED forever (long-lived infrastructure objects - exactly what +// the sweep's own comment says it favors), that population alone filled the +// sweep's 256-entry-per-pass cap on every single call, permanently starving +// any EXPANDED entry with a higher tag (e.g. a static field's collection +// node, admitted only once its owning class first loads, well after +// startup) of ever being re-queued - a bug proved directly, before the fix, +// by this same test (then named +// StaleExpandedRotationStarvesHighTagEntryBehindLowTagPopulation). +// +// _stale_expanded_rotation_cursor now makes collectStaleExpandedEntriesForRotation() +// resume from where the previous call left off instead of always restarting +// at tag 1, the same wrapping-cursor guarantee +// RotationCoversAllEntriesWithinCeilNOverR above already proves for +// collectStaleRootKindEntriesForRotation(): every entry, including one +// sitting behind an arbitrarily large low-tag population, gets a turn within +// ceil(table_size / max_count) calls. +// +// Driven directly against collectStaleExpandedEntriesForRotation() (the same +// unit-level style as RotationCoversAllEntriesWithinCeilNOverR above) rather +// than through a full JVMTI-mocked BFS walk: this property is intrinsic to +// the selection function's own tag-order scan, so it needs neither a real +// graph nor runPass()'s pacing/budget machinery to demonstrate. +TEST_F(ReferenceChainsBfsTest, StaleExpandedRotationCoversHighTagEntryBehindLowTagPopulationWithinBoundedPasses) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + const int lowTagBudget = ReferenceChainsTestAccessor::staleExpandedRotationBudget(); + // Comfortably above the 256-entry cap, so the low-tag population alone + // would fill every sweep before an always-from-1 scan could ever reach + // the high-tag entry below - mirrors a real multi-GiB heap's frontier + // table, which accumulates far more than 256 long-lived, perpetually- + // EXPANDED entries (bootstrap classes, caches, etc.) well before any one + // leak-candidate class even loads. + const int lowTagPopulation = lowTagBudget + 50; + for (jlong tag = 1; tag <= lowTagPopulation; tag++) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + } + + // The leak candidate's own owning node - e.g. LEAK_BUFFER's list, admitted + // via a static field only once its class loads, well after the JVM's own + // bootstrap population already occupies every low tag number. + const jlong highTag = lowTagPopulation + 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, highTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + // Several simulated passes: each iteration mirrors one real pass - + // collectStaleExpandedEntriesForRotation() runs once, then + // clearPriorityExpand() mirrors expandFrontier() having drained whatever + // it selected before the next pass's sweep resumes from the cursor. + const int table_size = lowTagPopulation + 1; + const int calls = (table_size + lowTagBudget - 1) / lowTagBudget; + bool highTagSelected = false; + std::unordered_set covered; + for (int pass = 0; pass < calls && !highTagSelected; pass++) { + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation( + lowTagBudget); + for (jlong tag : selected) { + covered.insert(tag); + if (tag == highTag) { + highTagSelected = true; + } + } + ReferenceChainsTestAccessor::clearPriorityExpand(); + } + + EXPECT_TRUE(highTagSelected) + << "highTag was never selected within ceil(table_size / max_count) " + "passes - the fix's coverage guarantee does not hold"; + EXPECT_EQ((size_t)table_size, covered.size()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// trackLeakAccumulation() - the admission-time hook (called from +// admitObject(), the single shared admission path) that aggregates by +// (leaf_class_tag, parent_class_tag) signature and by individual parent +// fanout - stable JVMTI class tags (classTagAllocator.h), not classMap +// dictionary ids, precisely because that dictionary can be compacted/ +// regenerated independently, silently reassigning the same class a +// different id at different times (found via the external-process test - +// see class_tag's own comment, referenceChains.h). Driven directly, pure +// FrontierTable/map logic with no JVMTI dependency of its own. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationAggregatesBySignatureAndFanout) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987; + constexpr u32 kParent1Klass = 100, kParent2Klass = 200; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong parent1Tag = 1, parent2Tag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parent1Tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParent1Klass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parent2Tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParent2Klass)); + + // 3 children of the watched leaf klass under parent1, 1 under parent2 - + // each call simulates one admission (the childTag argument is only used + // by production code for logging/future use, not read by this method). + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 10); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 11); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent1Tag, 12); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parent2Tag, 20); + + EXPECT_EQ(3u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParent1Klass)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParent2Klass)); + EXPECT_EQ(3u, ReferenceChainsTestAccessor::leakParentFanout(parent1Tag)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parent2Tag)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsUnwatchedKlass) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, /*class_tag=*/100)); + + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, /*class_tag=*/555, + parentTag, 10); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsRootAttachedChild) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); + + // parent_tag == 0 - a root-attached leaf itself, nothing to attribute a + // container to. + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, 987, /*parent_tag=*/0, 10); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationSkipsWhenParentNotFound) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({987}); + + // parent_tag=99 was never inserted - graceful no-op, not a crash. + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, 987, /*parent_tag=*/99, 10); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + + tracker->stop(); +} + +// Proof of the actual bug this design was found fixing: the classMap +// dictionary id (referrer_klass) for the exact same class can differ +// depending on which subsystem/generation resolved it (see class_tag's own +// comment, referenceChains.h, for the real-world case - "[B" resolving to +// two different classMap ids for LivenessTracker vs. ReferenceChainTracker). +// Matching must work via class_tag regardless of what referrer_klass says. +TEST_F(ReferenceChainsBfsTest, TrackLeakAccumulationMatchesByClassTagEvenWhenReferrerKlassDiffers) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafClassTag = 987; + constexpr u32 kParentClassTag = 100; + // Deliberately different, "wrong" classMap ids - simulating exactly the + // compaction/regeneration scenario that broke referrer_klass-based + // matching. If matching used referrer_klass at all, this test would fail. + constexpr u32 kParentStaleReferrerKlass = 555555; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafClassTag}); + + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, kParentStaleReferrerKlass, + kParentClassTag)); + + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafClassTag, + parentTag, 10); + + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafClassTag, + kParentClassTag)); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// collectLeakAccumulationCandidatesForRotation() - the two-tier design +// itself (see its own header comment, referenceChains.cpp, for the full +// rationale). Driven directly against the aggregation state +// trackLeakAccumulation() above populates, the same unit-level style as the +// other rotation collectors. +// --------------------------------------------------------------------------- + +// The central discriminating test for the whole design (per the "ubiquitous +// common leaf class held by many small unrelated parents" concern this +// design exists to solve): a signature with a LARGE but FLAT total (many +// unrelated parents, e.g. a common leaf class scattered across a real +// classpath) must NOT outrank a signature with a SMALLER but GROWING total +// (the actual leak) once a growth history exists - retained-size-style +// ranking alone would pick the wrong one every time. +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationPrioritizesGrowingSignatureOverLargeFlatOne) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987; + constexpr u32 kGrowingParentKlass = 100; // signature A: the real leak + constexpr u32 kUbiquitousParentKlass = 999; // signature B: common, but flat + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + // Signature A: one parent, growing. + jlong growingParentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, growingParentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kGrowingParentKlass)); + + // Signature B: 20 distinct, unrelated parents, each holding just 1-2 + // instances of the same common leaf klass - a much LARGER total than A, + // but it will not grow between passes. + constexpr int kUbiquitousParentCount = 20; + std::vector ubiquitousParentTags; + for (int i = 0; i < kUbiquitousParentCount; i++) { + jlong tag = 100 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kUbiquitousParentKlass)); + ubiquitousParentTags.push_back(tag); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, tag, 1000 + i); + } + // Pass 1: A has fanout 5, B has total 20 (20 parents x 1 each) - B is + // larger. First-ever call has no prior snapshot, so both deltas equal + // their totals; B legitimately wins this one call (nothing to compare + // growth against yet). + for (int i = 0; i < 5; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + growingParentTag, 2000 + i); + } + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( + ReferenceChainsTestAccessor::leakAccumulationRotationBudget()); + + // Pass 2: B stays exactly flat (no new admissions); A grows from 5 to 8. + for (int i = 0; i < 3; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + growingParentTag, 3000 + i); + } + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( + ReferenceChainsTestAccessor::leakAccumulationRotationBudget()); + + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ(growingParentTag, selected[0]) + << "the growing signature's parent must be selected, even though " + "the flat-but-larger signature has a much bigger absolute total"; + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationRanksByFanoutWithinWinningSignature) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong lowFanoutTag = 1, highFanoutTag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, lowFanoutTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, highFanoutTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, lowFanoutTag, 10); + for (int i = 0; i < 5; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + highFanoutTag, 20 + i); + } + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ(highFanoutTag, selected[0]) << "higher fanout ranks first"; + EXPECT_EQ(lowFanoutTag, selected[1]); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationRespectsMaxCountAndDedup) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong tag1 = 1, tag2 = 2, tag3 = 3; + for (jlong tag : {tag1, tag2, tag3}) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, tag, 10); + } + // tag2 already queued from an earlier collector this same pass - must + // be skipped even though it qualifies structurally. + ReferenceChainsTestAccessor::pushPriorityExpand(tag2); + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation( + /*max_count=*/1); + EXPECT_EQ(1u, selected.size()) << "capped at max_count"; + EXPECT_NE(tag2, selected[0]) << "already-queued tag must not be re-selected"; + + tracker->stop(); +} + +// Reversed on round-4 pod evidence (ev-leaktag-onpod-round4): the previous +// EXPANDED-only selection made the targeted tier select ZERO every pass +// on a live leak - the growing holders are un-expanded FRONTIER-state +// backlog entries that the starved pending lane never reaches (a 127k +// backlog at ~120-200 objects/min). A FRONTIER-state parent must be +// selected AND placed at the head of the priority lane so the next +// expandFrontier() batch reaches it ahead of the stale re-walks already +// queued (the same pod's priority deque held ~1016 entries). +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationSelectsUnexpandedFrontierParentAheadOfBacklog) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + // Stale re-walks already sitting in the priority lane (push_back, as + // the other two collectors do). + ReferenceChainsTestAccessor::pushPriorityExpand(900); + ReferenceChainsTestAccessor::pushPriorityExpand(901); + + jlong notYetExpandedTag = 1, expandedLowFanoutTag = 2; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, notYetExpandedTag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, expandedLowFanoutTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + for (int i = 0; i < 10; i++) { + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + notYetExpandedTag, 10 + i); + } + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, + expandedLowFanoutTag, 100); + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ(notYetExpandedTag, selected[0]) + << "the FRONTIER-state parent qualifies and outranks the " + "lower-fanout EXPANDED one"; + EXPECT_EQ(expandedLowFanoutTag, selected[1]); + + std::vector queue = ReferenceChainsTestAccessor::priorityExpandContents(); + ASSERT_GE(queue.size(), 4u); + EXPECT_EQ(notYetExpandedTag, queue[0]) + << "the targeted un-expanded holder must JUMP the backlog, not " + "queue behind the stale re-walks"; + EXPECT_EQ(expandedLowFanoutTag, queue[1]) + << "selection order must be preserved at the head (fanout rank)"; + EXPECT_EQ(900, queue[2]); + EXPECT_EQ(901, queue[3]); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationReturnsEmptyWhenNothingHasGrownSincePreviousPass) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parentTag, 10); + + // First call establishes the baseline (delta == total, since there is no + // prior snapshot) and selects it. + std::vector firstPass = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + ASSERT_EQ(1u, firstPass.size()); + ReferenceChainsTestAccessor::clearPriorityExpand(); + + // Second call, nothing new admitted - delta is now 0 for every + // signature, so nothing should be selected. + std::vector secondPass = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + EXPECT_TRUE(secondPass.empty()) + << "no signature grew since the previous pass's snapshot"; + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, LeakAccumulationRotationReturnsEmptyWhenNoSignaturesTracked) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + std::vector selected = + ReferenceChainsTestAccessor::collectLeakAccumulationCandidatesForRotation(10); + EXPECT_TRUE(selected.empty()); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// seedLeakAccumulationForNewlyWatchedKlass() - the cold-start fix: retroactively +// seeds the aggregation from entries ALREADY admitted before a klass_id +// started being watched, since trackLeakAccumulation() alone only ever sees +// admissions happening after watching starts, and the container that +// actually needs re-expansion is typically already fully admitted by then +// (found via the external-process test: the delegate ArrayList sits one hop +// below the root-attached wrapper, and admitStaticFieldRoots()'s own sweep +// admits both in the same call, long before any leak signal can plausibly +// have fired). +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationPopulatesFromAlreadyAdmittedEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + // Entries inserted directly (as if admitted by an earlier pass), with no + // watched klass_id set at all yet at insertion time - trackLeakAccumulation() + // was never called for any of these. + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + for (int i = 0; i < 4; i++) { + jlong childTag = 10 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, childTag, parentTag, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); + } + ASSERT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()) + << "nothing tracked yet - trackLeakAccumulation() was never called"; + + ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); + + EXPECT_EQ(4u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParentKlass)); + EXPECT_EQ(4u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationSkipsNonMatchingAndNonExpandedEntries) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kOtherKlass = 555, kParentKlass = 100; + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + // Wrong class - must not be counted. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 10, parentTag, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0, /*referrer_klass=*/0, kOtherKlass)); + // Right class, but still FRONTIER (not yet EXPANDED) - must not be counted. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 11, parentTag, 1, FrontierEntryState::FRONTIER, + /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); + // Right class, root-attached (no real parent) - must not be counted. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 12, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kLeafKlass)); + + ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); + + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakSignatureCount()); + + tracker->stop(); +} + +TEST_F(ReferenceChainsBfsTest, SeedLeakAccumulationComposesWithOngoingIncrementalUpdates) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987, kParentKlass = 100; + jlong parentTag = 1; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, parentTag, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, kParentKlass)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 10, parentTag, 1, FrontierEntryState::EXPANDED, + /*root_kind=*/0, /*referrer_klass=*/0, kLeafKlass)); + + // Retroactive seed sees the one pre-existing child. + ReferenceChainsTestAccessor::seedLeakAccumulationForNewlyWatchedKlass(kLeafKlass); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + + // A genuinely new admission after watching starts must add on top of the + // retroactive baseline, not reset or double it. + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, parentTag, 11); + + EXPECT_EQ(2u, ReferenceChainsTestAccessor::leakParentFanout(parentTag)); + EXPECT_EQ(2u, ReferenceChainsTestAccessor::leakSignatureTotal(kLeafKlass, kParentKlass)); + + tracker->stop(); +} + +// --------------------------------------------------------------------------- +// Smoke test simulating the hotdog pod conditions that starved BFS: +// +// 1. GetObjectsWithTags quadratic bottleneck: a large frontier backlog +// makes each GetObjectsWithTags call expensive. The self-calibrating +// adaptive batch_size must keep it bounded. +// +// 2. Shared deadline bug: the static-field sweep's FollowReferences ate +// the entire per-pass wall-clock deadline, leaving expand with zero +// time. The deadline split gives each sub-operation its own fresh +// deadline. +// +// 3. Rolling resume: when FollowReferences truncates mid-batch (budget +// exhausted), the fully-processed entries must be popped (mark +// EXPANDED) and only the partially-processed + unvisited entries left +// for retry. +// +// This test builds a graph with a static-field root leading to a chain of +// objects (simulating the leaking collection), plus a small set of +// distractor roots to keep the search RUNNING. It runs passes with a +// small budget so expand truncates mid-batch, then verifies the rolling +// resume and progress properties. +// --------------------------------------------------------------------------- + +TEST_F(ReferenceChainsBfsTest, RollingResumePopsProcessedEntriesOnTruncatedBatch) { + Arguments args; + // budget=4: small enough that expand truncates mid-batch after admitting + // a few children. firstpassbudget=1000: large enough to enumerate all + // roots in the first pass without truncating root enum. + ASSERT_FALSE(args.parse( + "referencechains=true:hops=5000:budget=4:firstpassbudget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // A static-field root: classNode -> listNode (the leaking collection). + int classNode = addNode(); + int listNode = addNode(); + + // A chain of 20 children hanging off listNode. With budget=4, the + // callback admits 4 children then returns JVMTI_VISIT_ABORT + // (BUDGET_EXHAUSTED), truncating mid-batch. + constexpr int kChainLen = 20; + std::vector chainNodes(kChainLen); + for (int i = 0; i < kChainLen; i++) { + chainNodes[i] = addNode(); + } + + // Distractor roots: 20 independent JNI-global roots, each with one child. + // Enough to keep the search RUNNING but small enough to drain quickly. + constexpr int kDistractors = 20; + std::vector distractorRoots(kDistractors); + std::vector distractorChildren(kDistractors); + for (int i = 0; i < kDistractors; i++) { + distractorRoots[i] = addNode(); + distractorChildren[i] = addNode(); + } + + // addClass() must come after all addNode() calls. + addClass((void *)&node_tags[classNode], "Lcom/rc/SmokeTestHolder;"); + + script = { + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, listNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, listNode, chainNodes[0], -1}, + }; + for (int i = 0; i + 1 < kChainLen; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, chainNodes[i], chainNodes[i + 1], -1}); + } + for (int i = 0; i < kDistractors; i++) { + script.push_back({JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, distractorRoots[i], -1}); + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, distractorRoots[i], distractorChildren[i], -1}); + } + + // Phase 1: run passes until listNode is admitted via the static-field sweep. + bool truncated = true; + jlong listTag = 0; + for (int i = 0; i < 200 && listTag == 0; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + listTag = tags_ever_assigned[listNode]; + } + ASSERT_NE(0, listTag) << "listNode was never admitted to the frontier"; + + // Phase 2: run passes until listNode is expanded (rolling resume pops it). + for (int i = 0; i < 200; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + FrontierEntry entry{}; + if (frontier->lookup(listTag, &entry) && + entry.state == FrontierEntryState::EXPANDED) { + break; + } + } + FrontierEntry listEntry{}; + ASSERT_TRUE(frontier->lookup(listTag, &listEntry)); + EXPECT_EQ(FrontierEntryState::EXPANDED, listEntry.state) + << "listNode should be EXPANDED after rolling resume popped it"; + + // Verify some chain children were admitted. + int admittedChildren = 0; + for (int i = 0; i < kChainLen; i++) { + if (tags_ever_assigned[chainNodes[i]] != 0) admittedChildren++; + } + EXPECT_GT(admittedChildren, 0) + << "No chain children were admitted — expand never ran"; + + // Phase 3: run more passes until all chain children are admitted. + for (int i = 0; i < 500 && admittedChildren < kChainLen; i++) { + ASSERT_EQ(SearchState::RUNNING, tracker->searchState()); + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + admittedChildren = 0; + for (int j = 0; j < kChainLen; j++) { + if (tags_ever_assigned[chainNodes[j]] != 0) admittedChildren++; + } + } + EXPECT_EQ(kChainLen, admittedChildren) + << "Not all chain children were admitted within bounded passes"; + + tracker->stop(); +} + +// Verify the AIMD adaptive batch_size: with the per-call EMA over the CPU +// budget, expandFrontier should multiplicatively decrease the batch; under +// the budget it should additively increase toward the cap. The mock +// GetObjectsWithTags is instant (no real tag-map cost), so we drive the EMA +// by hand and verify the AIMD response, not the timing. +TEST_F(ReferenceChainsBfsTest, AdaptiveBatchSizeProportionalToWindow) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Adaptive-batch state is zeroed by reset() (SetUp) but zeroed here too + // for the same reason as before: exact per-phase arithmetic below. + ReferenceChainsTestAccessor::setGotwEmaCallNs(0); + ReferenceChainsTestAccessor::setGotwBatchSize(0); + ReferenceChainsTestAccessor::setPassDeadlineNs(0); + + // Seed a frontier root manually (mirrors PollWatchedTargetsTest's + // seeding style): node carries frontier tag 1, pending expansion has + // exactly that tag. No runPass() - a pass would drain the tiny graph to + // COMPLETED and release all tags, and its rotation phase adds extra + // GetObjectsWithTags calls, both of which break per-call arithmetic. + // The script stays empty until the admission-sanity phase below, so + // each expandFrontier() drive runs exactly one batch (one control + // update) and admits nothing. + int rootNode = addNode(); + int childNode = addNode(); + node_tags[rootNode] = 1; + ASSERT_TRUE(tracker->frontierTable()->insert( + 1, 0, 1, 0, FrontierEntryState::EDGE)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + int edges = 0; + const u64 budget = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); + + // --- Populate phase: first GetObjectsWithTags call. The EMA should be + // non-zero afterwards, and the near-zero mock call time means the + // window (nominal budget, no deadline) fits ~unbounded many calls - + // the proportion scales the batch all the way to the cap. + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_NE(0u, ReferenceChainsTestAccessor::gotwEmaCallNs()) + << "per-call EMA should be populated after first GetObjectsWithTags"; + EXPECT_EQ(ReferenceChainsTestAccessor::gotwMaxBatch(), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "near-free call should scale the batch to the cap"; + + // --- Shrink phase: EMA at 2x the window with no deadline -> batch + // halves (512 x 1 / 1.6 after the EMA update). + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget * 2); + ReferenceChainsTestAccessor::setGotwBatchSize(512); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + // EMA after the call: 2x window x 0.8 + mock elapsed/5 - slightly + // above 1.6x window, so the exact expectation is computed from the + // actual EMA the same way the control law does (window = nominal + // budget, no deadline): next = 512 x window / ema. + EXPECT_EQ((size_t)(512ULL * budget / + std::max(ReferenceChainsTestAccessor::gotwEmaCallNs(), + 1ULL)), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "EMA at ~1.6x the window should scale the batch to 512/1.6"; + + // --- Grow phase: EMA at half the window -> batch scales up 2.5x, + // i.e. the floor-dominated regime GROWS the batch (the whole point of + // the proportional law - the old AIMD could not grow past a fixed + // budget even when bigger batches were nearly free). + ReferenceChainsTestAccessor::setGotwBatchSize(64); + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget / 2); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + // Same computation from the actual post-call EMA (~0.4x window): + // next = 64 x window / ema. + EXPECT_EQ((size_t)(64ULL * budget / + std::max(ReferenceChainsTestAccessor::gotwEmaCallNs(), + 1ULL)), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "EMA under the window should scale the batch up proportionally"; + + // --- Deadline-window phase: with a live pass deadline the window is the + // REMAINING time, not the nominal budget. A deadline 10x the budget out + // with EMA ~ 1x budget scales the batch 10x/0.8 - past the cap, so the + // clamp holds it at GOTW_MAX_BATCH (robust to the nanoseconds the call + // itself consumes). + ReferenceChainsTestAccessor::setPassDeadlineNs( + OS::nanotime() + budget * 10); + ReferenceChainsTestAccessor::setGotwBatchSize(64); + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_EQ(ReferenceChainsTestAccessor::gotwMaxBatch(), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "a wide remaining deadline should grow the batch to the cap"; + ReferenceChainsTestAccessor::setPassDeadlineNs(0); + + // --- Admission sanity: expansion still walks the graph. Root -> child + // edge, one more drive, child must be admitted. + script.push_back({JVMTI_HEAP_REFERENCE_FIELD, rootNode, childNode, -1}); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_NE(0, tags_ever_assigned[childNode]) + << "expandFrontier failed to admit childNode with adaptive batch_size"; + + tracker->stop(); +} + +// gotwWindowNs() backlog-pressure widening, unit level: the pod regime is +// a remaining pass window (~10ms) smaller than the measured per-call floor +// (~22-40ms at a 242k-entry tag map), against a lane 127k deep. In that +// regime the previous window math handed the proportional law a window +// smaller than one unavoidable call, collapsing the batch to +// GOTW_MIN_BATCH forever (batch=8 live on every call while the backlog +// drained at ~120-200 objects/min). The widening makes the law size the +// batch UP instead - but ONLY under real backlog depth, so rotation +// fast-lane batches stay deadline-sized. +TEST_F(ReferenceChainsBfsTest, GotwWindowWidensOnlyUnderBacklogPressure) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + const u64 budget = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); + const size_t depth = ReferenceChainsTestAccessor::gotwBacklogMinDepth(); + const u64 mult = ReferenceChainsTestAccessor::gotwBacklogWindowMult(); + const u64 floor = budget * 2; // any floor above the nominal window + + // No deadline and no EMA yet: the nominal budget window. + ReferenceChainsTestAccessor::setGotwEmaCallNs(0); + EXPECT_EQ(budget, ReferenceChainsTestAccessor::gotwWindowNs(0, depth)); + + ReferenceChainsTestAccessor::setGotwEmaCallNs(floor); + + // Floor above the remaining window but a SHALLOW lane: no widening - + // the remaining window stands (rotation fast-lane stays cheap). + EXPECT_EQ(1u, ReferenceChainsTestAccessor::gotwWindowNs(1, 1)); + + // Floor above the remaining window and a DEEP lane: widened to + // EMA x mult, never below the remaining window itself. + EXPECT_EQ(floor * mult, + ReferenceChainsTestAccessor::gotwWindowNs(1, depth)); + + // Floor BELOW the remaining window: no widening even at depth - + // the ordinary proportional law already fits the call in the window. + ReferenceChainsTestAccessor::setGotwEmaCallNs(budget / 2); + EXPECT_EQ(budget, + ReferenceChainsTestAccessor::gotwWindowNs(budget, depth)); + ReferenceChainsTestAccessor::setGotwEmaCallNs(floor); + + // Floor above the NOMINAL window (deadline already passed, the exact + // pod's post-call state) at depth: still widened - the floor is paid + // by the next call regardless, so the batch must amortize it. + EXPECT_EQ(floor * mult, + ReferenceChainsTestAccessor::gotwWindowNs(0, depth)); + + tracker->stop(); +} + +// The widened window in action through the real control loop: one +// GetObjectsWithTags call whose floor (simulated by the mock's busy-wait) +// exceeds both the remaining pass deadline and the nominal window, with +// a backlog deeper than GOTW_BACKLOG_MIN_DEPTH, must GROW the calibrated +// batch (calib x mult, exactly - the window scales with the measured EMA) +// instead of collapsing it to GOTW_MIN_BATCH. This is the round-4 pod +// failure reproduced mechanically: pre-widening, next = calib x nominal / +// ema < calib clamps to MIN forever. +TEST_F(ReferenceChainsBfsTest, AdaptiveBatchGrowsWhenFloorExceedsWindowUnderDeepBacklog) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // A pass deadline a fraction of the simulated per-call floor: the call + // overruns it (exactly the pod's 10ms window vs 22-40ms floor), so after + // the call the loop's deadline check stops the invocation with ONE + // control update - deterministic arithmetic for the assertion below. + gotw_delay_ns = ReferenceChainsTestAccessor::gotwCpuBudgetNs(); // 25ms floor + ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 5000000ULL); + ReferenceChainsTestAccessor::setGotwBatchSize(ReferenceChainsTestAccessor::gotwMinBatch()); + ReferenceChainsTestAccessor::setGotwEmaCallNs(0); // seeded by the call below + + // A pending lane deep enough to cross GOTW_BACKLOG_MIN_DEPTH. The tags + // are unresolvable (never assigned to a node): each batch resolves + // nothing, which is fine - this test drives the control law, not + // admissions. + const size_t depth = ReferenceChainsTestAccessor::gotwBacklogMinDepth() + 1; + for (size_t i = 0; i < depth; i++) { + ReferenceChainsTestAccessor::pushPendingExpandForTest( + (jlong)(1000000 + i)); + } + + int edges = 0; + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + + // EMA after the call = the busy-wait floor (~25ms). The pass deadline is + // long past, so the window is widened to EMA x GOTW_BACKLOG_WINDOW_MULT, + // and the control law computes calib x window / ema = calib x mult - + // exactly, because the window is a whole multiple of the same EMA it + // divides by. + EXPECT_EQ(ReferenceChainsTestAccessor::gotwMinBatch() * + ReferenceChainsTestAccessor::gotwBacklogWindowMult(), + ReferenceChainsTestAccessor::gotwBatchSize()) + << "the floor-dominated deep-backlog regime must GROW the batch, " + "not clamp it to GOTW_MIN_BATCH"; + + ReferenceChainsTestAccessor::setPassDeadlineNs(0); + gotw_delay_ns = 0; + tracker->stop(); +} + +// FAIR-SHARE DRAIN persistence: the lane toggle must survive across +// expandFrontier() invocations. With per-invocation deadlines bounding an +// invocation to a single batch (the production regime: one ~25-30ms +// GetObjectsWithTags call of a 50ms window), a per-invocation reset to +// "priority first" made priority win EVERY invocation and the ordinary +// pending lane was never drained - observed live on hotdog (pending grew +// 109k->113k over 260 passes, every call edges=0 stale re-walks). Here: +// invocation 1 drains the priority lane, the toggle flips to pending; +// priority is refilled (rotation would) and invocation 2 must still drain +// the PENDING lane despite priority being non-empty. +TEST_F(ReferenceChainsBfsTest, FairShareLaneAlternationPersistsAcrossInvocations) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int rootNode = addNode(); + int otherRoot = addNode(); + // Two live boundary objects: tag 1 in pending, tag 2 in priority. + node_tags[rootNode] = 1; + node_tags[otherRoot] = 2; + ASSERT_TRUE(tracker->frontierTable()->insert( + 1, 0, 1, 0, FrontierEntryState::EDGE)); + ASSERT_TRUE(tracker->frontierTable()->insert( + 2, 0, 1, 0, FrontierEntryState::EDGE)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(1); + ReferenceChainsTestAccessor::pushPriorityExpand(2); + int edges = 0; + + // Mock GetObjectsWithTags calls are ~free, so without a deadline a + // single expandFrontier() invocation would drain BOTH lanes in one + // loop. A delayed mock call (gotw_delay_ns below) plus a deadline set + // to a fraction of that delay bounds each invocation to exactly ONE + // batch: the first iteration's top-of-loop check passes (the deadline + // is ~200us away), the delayed call burns past it, and the second + // iteration's check breaks - the production regime, where one real + // ~25-30ms call consumes a 50ms window. + gotw_delay_ns = 1 * 1000 * 1000; // 1ms + ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 200 * 1000); + + // Invocation 1: priority first (the standing preference). + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::priorityExpandSize()) + << "first invocation should drain the priority lane"; + EXPECT_EQ(1u, ReferenceChainsTestAccessor::pendingExpandSize()) + << "first invocation must leave the pending lane for the next one"; + EXPECT_FALSE(ReferenceChainsTestAccessor::expandLanePreferPriority()); + + // Rotation refills the priority lane; invocation 2 must STILL prefer + // the pending lane - the toggle persists, it is not reset per call. + ReferenceChainsTestAccessor::pushPriorityExpand(2); + ReferenceChainsTestAccessor::setPassDeadlineNs(OS::nanotime() + 200 * 1000); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, + &mock_jni, &edges); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::pendingExpandSize()) + << "second invocation should drain the pending lane"; + EXPECT_EQ(1u, ReferenceChainsTestAccessor::priorityExpandSize()) + << "second invocation must leave the refilled priority lane alone"; + EXPECT_TRUE(ReferenceChainsTestAccessor::expandLanePreferPriority()); + + tracker->stop(); +} +// FANOUT HYGIENE: a _leak_parent_fanout entry whose parent no longer +// resolves in the frontier (pruned: dead object, or a search-restart wipe) +// can never be re-walked, so collectStaleExpandedEntriesForRotation() must +// erase it during selection rather than skip it forever - without the +// erase, the fanout grows monotonically with corpses (observed live at +// ~11k entries of overwhelmingly-dead old backing arrays), which both +// bloats the selection scan and turns the fanout cursor's lap arithmetic +// into mostly wasted skips. +TEST_F(ReferenceChainsBfsTest, StaleRotationEvictsDeadFanoutParents) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + constexpr u32 kLeafKlass = 987; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + FrontierTable *frontier = tracker->frontierTable(); + + // Live fanout parent 1 and dead fanout parent 5 (frontier entry pruned). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, + /*class_tag=*/42)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, + /*class_tag=*/43)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 1, 10); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 5, 20); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(1)); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(5)); + + frontier->clear(5); // parent 5's object died / search restart pruned it + + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(4); + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::leakParentFanout(5)) + << "dead fanout parent must be erased during selection"; + EXPECT_EQ(1u, ReferenceChainsTestAccessor::leakParentFanout(1)) + << "live fanout parent must survive"; + + tracker->stop(); +} + + +// Leak-tag interception (design A + C): an object pre-tagged with a leak tag +// (as LivenessTracker::tagLeakInstances() would have set on a tracked leaking +// instance) must be admitted by converting the leak tag to a frontier tag, +// with the leak tag preserved in the frontier entry so buildChainEvent() +// emits it as target_tag - the ReferenceChain <-> HeapLiveObject correlation +// key. An untagged sibling of the same class must get an ordinary admit with +// no leak tag. +TEST_F(ReferenceChainsBfsTest, LeakTagInterceptionConvertsToFrontierTagAndCorrelates) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + + int rootNode = addNode(); + int leakChild = addNode(); + int plainChild = addNode(); + // Simulate tagLeakInstances(): the tracked leaking instance already + // carries a leak tag; the sibling does not. + node_tags[leakChild] = leak_tag; + script = { + {JVMTI_HEAP_REFERENCE_JNI_GLOBAL, -1, rootNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, rootNode, leakChild, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, rootNode, plainChild, -1}, + }; + + // A single pass drains this tiny graph to completion, and a completed + // search releases all JVMTI tags (releaseSearchTags(), "tagsReleased" + // in runPass's own log) - so read the tags from tags_ever_assigned, + // which records each tag at assignment time and is never reset (see + // its own comment), not from node_tags (which reads 0 after release). + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + + // The leak-tagged child's tag was REPLACED by a frontier tag (small + // positive, outside the leak range). + jlong leak_ftag = tags_ever_assigned[leakChild]; + ASSERT_NE(leak_tag, leak_ftag) + << "leak tag was never intercepted - BFS did not reach the object"; + ASSERT_GT(leak_ftag, 0); + EXPECT_LT(leak_ftag, leak_tag) << "frontier tag must be outside leak range"; + + // The frontier entry preserves the leak tag for correlation. + EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(leak_ftag)); + + // The untagged sibling got an ordinary admit: frontier tag assigned, but + // no leak tag stored. + jlong plain_ftag = tags_ever_assigned[plainChild]; + ASSERT_GT(plain_ftag, 0); + EXPECT_EQ(0, ReferenceChainsTestAccessor::frontierLeakTag(plain_ftag)); + + // Design C: buildChainEvent() reports the leak tag as target_tag for the + // leak-tagged instance, and the plain frontier tag for the sibling. + ReferenceChainEvent event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, leak_ftag, &event)); + EXPECT_EQ((u64)leak_tag, event._target_tag) + << "chain target tag must be the leak tag (correlation key)"; + EXPECT_GE(event._depth, 1u) << "leak child sits behind the root, not at it"; + + ReferenceChainEvent plain_event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, plain_ftag, &plain_event)); + EXPECT_EQ((u64)plain_ftag, plain_event._target_tag) + << "untagged instance must keep the frontier tag as target tag"; + + tracker->stop(); +} + +// Candidate-scoped reach, prong 1 (walkCandidateThreadLocals()): a leak +// held through the leaking thread's ThreadLocalMap must be intercepted +// with its full chain by ONE bounded walk from the Thread object, no +// matter what the ordinary BFS backlog state is - and the walk's gates +// must keep it off the Thread's non-thread-local fields entirely. +TEST_F(ReferenceChainsBfsTest, ThreadWalkDescendsOnlyThreadLocalMapAndInterceptsLeak) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // The classes descendFromAnchor()'s resolutions look up by FindClass + // name (see registerClassForFindClass' own comment) + the scripted + // graph's own classes. resolveLoadedClasses() below tags every + // registered class, which is what makes the class-tag gates resolvable. + void *threadCls = (void *)0x3001, *tlmapCls = (void *)0x3002, + *loaderCls = (void *)0x3003, *holderCls = (void *)0x3004, + *chunkCls = (void *)0x3005; + int tlmapIdx = + registerClassForFindClass(tlmapCls, + "java/lang/ThreadLocal$ThreadLocalMap", + "Ljava/lang/ThreadLocal$ThreadLocalMap;"); + int loaderIdx = + registerClassForFindClass(loaderCls, "java/lang/ClassLoader", + "Ljava/lang/ClassLoader;"); + registerClassForFindClass(threadCls, "java/lang/Thread", + "Ljava/lang/Thread;"); + int holder = addClass(holderCls, "Lcom/rc/descendwalk/Holder;"); + int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/LeakChunk;"); + thread_class = threadCls; + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + + int threadNode = addNode(); + int threadNode2 = addNode(); // second candidate thread: fresh-admission path + int tlmapNode = addNode(); + int loaderNode = addNode(); // Thread's contextClassLoader: anchor gate + int loaderNode2 = addNode(); // a ClassLoader below the gate: no-descend + int holderNode = addNode(); + int leakChunk = addNode(); + + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + node_tags[leakChunk] = leak_tag; + + // Topological order (mock_FollowReferences replays edges in script + // order, expanding only refs the production callback said to descend + // into). + script = { + {JVMTI_HEAP_REFERENCE_FIELD, threadNode, tlmapNode, + /*class_idx=*/-1}, + {JVMTI_HEAP_REFERENCE_FIELD, threadNode, loaderNode, + /*class_idx=*/-1}, + {JVMTI_HEAP_REFERENCE_FIELD, tlmapNode, holderNode, holder}, + {JVMTI_HEAP_REFERENCE_FIELD, tlmapNode, loaderNode2, + /*class_idx=*/-1}, + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, leakChunk, chunk}, + }; + // The anchor gate compares the REFEREE's class tag, so the thread + // edges' class_idx values matter: the tlmap edge carries + // ThreadLocalMap's tag, and the loader edges ClassLoader's. + script[0].class_idx = tlmapIdx; + script[1].class_idx = loaderIdx; + script[3].class_idx = loaderIdx; + + // The first thread walks the REUSE path: its Thread object is already + // admitted (root-attached THREAD entry + JVMTI tag) exactly as it is + // in production after the first walk pass or root enumeration. The + // mock keeps the scripted tag array (node_tags, what callbacks see as + // tag_ptr) separate from the pointer-keyed tag map (what GetTag/SetTag + // see), so the pre-anchored tag must be mirrored into both. + FrontierTable *frontier = tracker->frontierTable(); + jlong anchor_tag = + tracker->tagObject(&mock_jvmti, + reinterpret_cast(&node_tags[threadNode])); + ASSERT_GT(anchor_tag, 0); + node_tags[threadNode] = anchor_tag; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, anchor_tag, 0, 0, FrontierEntryState::FRONTIER, + (u8)JVMTI_HEAP_REFERENCE_THREAD)); + + // Two candidate slots, two qualifying tids: tid 777's thread is + // pre-anchored (reuse path), tid 778's is untagged (fresh-admission + // path - GetObjectClass + tagObject + insert, the path a never-walked + // thread takes in production). + jint tids0[] = {777}; + jint tids1[] = {778}; + ReferenceChainsTestAccessor::seedCandidateSlotForTest( + /*slot=*/0, /*klass_id=*/6, tids0, 1); + ReferenceChainsTestAccessor::seedCandidateSlotForTest( + /*slot=*/1, /*klass_id=*/6, tids1, 1); + tracker->registerThreadObject( + &mock_jni, 777, reinterpret_cast(&node_tags[threadNode])); + tracker->registerThreadObject( + &mock_jni, 778, reinterpret_cast(&node_tags[threadNode2])); + + int edges = 0; + ReferenceChainsTestAccessor::walkCandidateThreadLocalsForTest( + &mock_jvmti, &mock_jni, 1000, &edges); + + // The ThreadLocalMap-held chain was admitted end-to-end and the + // leak-tagged chunk was intercepted (tag replaced by a frontier tag, + // leak tag preserved for correlation). + jlong thread_ftag = anchor_tag; + jlong tlmap_ftag = tags_ever_assigned[tlmapNode]; + ASSERT_GT(tlmap_ftag, 0) << "anchor gate did not descend into ThreadLocalMap"; + jlong holder_ftag = tags_ever_assigned[holderNode]; + ASSERT_GT(holder_ftag, 0) << "walk did not descend below ThreadLocalMap"; + jlong chunk_ftag = tags_ever_assigned[leakChunk]; + ASSERT_NE(chunk_ftag, leak_tag) + << "leak-tagged chunk under the ThreadLocalMap was never intercepted"; + ASSERT_GT(chunk_ftag, 0); + EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); + + // Chain shape: Thread (root-attached, THREAD root kind) -> ThreadLocalMap + // -> holder -> chunk. + FrontierEntry thread_entry{}; + ASSERT_TRUE(frontier->lookup(thread_ftag, &thread_entry)); + EXPECT_EQ(0, thread_entry.parent_tag); + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_THREAD, thread_entry.root_kind); + FrontierEntry chunk_entry{}; + ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); + EXPECT_EQ(holder_ftag, chunk_entry.parent_tag); + EXPECT_EQ(3u, chunk_entry.depth); + + // The gates kept the walk off the metadata branches: neither the + // Thread's own contextClassLoader edge (anchor gate) nor a ClassLoader + // below ThreadLocalMap (no-descend gate) was admitted. + EXPECT_EQ(0, tags_ever_assigned[loaderNode]) + << "anchor gate must not admit the Thread's non-ThreadLocalMap fields"; + EXPECT_EQ(0, tags_ever_assigned[loaderNode2]) + << "no-descend gate must not admit fat-metadata classes below the anchor"; + + // The second thread took the fresh-admission path (no prior tag/entry): + // its Thread object was admitted root-attached with the THREAD root + // kind. (Its scripted tag array entry was never set, so the minted tag + // is read back through the pointer-keyed tag map via getTag.) + jlong thread2_ftag = ReferenceChainsTestAccessor::getTagForTest( + &mock_jvmti, reinterpret_cast(&node_tags[threadNode2])); + ASSERT_GT(thread2_ftag, 0) << "fresh thread anchor was never admitted"; + FrontierEntry thread2_entry{}; + ASSERT_TRUE(frontier->lookup(thread2_ftag, &thread2_entry)); + EXPECT_EQ(0, thread2_entry.parent_tag); + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_THREAD, thread2_entry.root_kind); + + tracker->stop(); +} + +// unregisterThreadObject() must defer the global-ref deletion to +// releaseEndedThreadRefs(): walkCandidateThreadLocals() copies the jobject +// out of _thread_objects under _thread_objects_lock, releases the lock, and +// can still be using it as a FollowReferences anchor when a concurrent +// ThreadEnd erases the entry - deleting there would be JNI use-after-free +// (see _thread_refs_pending_delete's comment). +TEST_F(ReferenceChainsBfsTest, ThreadRefUnregisterDefersGlobalRefDeletion) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int threadNode = addNode(); + tracker->registerThreadObject( + &mock_jni, 555, reinterpret_cast(&node_tags[threadNode])); + tracker->unregisterThreadObject(&mock_jni, 555); + // The erasing side only enqueues - no DeleteGlobalRef yet. + EXPECT_EQ(0, global_refs_deleted_); + + // The drain deletes exactly the queued ref, and draining an empty list + // is a no-op. + tracker->releaseEndedThreadRefs(&mock_jni); + EXPECT_EQ(1, global_refs_deleted_); + tracker->releaseEndedThreadRefs(&mock_jni); + EXPECT_EQ(1, global_refs_deleted_); + + tracker->stop(); +} + +// Candidate-scoped reach, prong 2 (collectStaticFieldAnchorsForRotation()/ +// walkStaticFieldAnchors()): the collector selects exactly the root-attached +// static-holder entries with a wrapping cursor, and the walk reaches a leak +// held 3-4 hops inside a static collection in one bounded call - the shape +// the one-hop Tier-2 rotation demonstrably cannot reach from an un-expanded +// FRONTIER holder on a rising heap (pod rounds 5-6). +TEST_F(ReferenceChainsBfsTest, StaticAnchorRotationWalksRootAttachedStaticHolders) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x4001, *chunkCls = (void *)0x4002; + addClass(holderCls, "Lcom/rc/descendwalk/StaticHolder;"); + int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/StaticChunk;"); + + int holderNode = addNode(); + int tableNode = addNode(); + int entryNode = addNode(); + int leakChunk = addNode(); + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + node_tags[leakChunk] = leak_tag; + + // Static Map -> table -> Entry -> chunk: the collection-shaped static + // holder's internals, deeper than one hop. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, tableNode, -1}, + {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, entryNode, leakChunk, chunk}, + }; + + // Seed the frontier exactly as the static sweep admits a static field's + // value: root-attached, STATIC_FIELD root kind, FRONTIER state. Plus a + // second durable-root anchor (JNI_GLOBAL - the pod-round-7 filter + // extension), a TRANSIENT-root decoy the collector must skip, and a + // chain-attached child (table entries only - deliberately NOT mirrored + // into node_tags, so the walk below freshly admits those nodes instead + // of tripping ALREADY_ADMITTED on stale scripted tags). + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 101; // mock_GetObjectsWithTags' resolvable tag + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 101, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + 101, 9001, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 102, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 103, 101, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_JNI_GLOBAL)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + 104, 9004, JVMTI_HEAP_REFERENCE_JNI_GLOBAL); + + std::vector selected = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 4); + // Both DURABLE root kinds are selected (STATIC_FIELD 101, JNI_GLOBAL + // 104, in cursor/tag order); the transient-root decoy and the child are + // not. The walk below drives only 101 (the resolvable node). + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ(101, selected[0]); + EXPECT_EQ(104, selected[1]); + std::vector walk_selected = {selected[0]}; + + // The static anchor's whole internal structure is admitted by one + // bounded walk, intercepting the leak tag at depth 3 below the holder. + int edges = 0; + ReferenceChainsTestAccessor::walkStaticFieldAnchorsForTest( + &mock_jvmti, &mock_jni, walk_selected, 1000, &edges); + jlong table_ftag = tags_ever_assigned[tableNode]; + jlong entry_ftag = tags_ever_assigned[entryNode]; + jlong chunk_ftag = tags_ever_assigned[leakChunk]; + ASSERT_GT(table_ftag, 0) << "table array was not reached by the anchor walk"; + ASSERT_GT(entry_ftag, 0) << "Entry was not reached one hop below table"; + ASSERT_NE(chunk_ftag, leak_tag) + << "leak-tagged chunk inside the static holder was never intercepted"; + EXPECT_EQ(leak_tag, ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); + FrontierEntry chunk_entry{}; + ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); + EXPECT_EQ(entry_ftag, chunk_entry.parent_tag); + EXPECT_EQ(3u, chunk_entry.depth); + + tracker->stop(); +} + +// Round-14 tiered selection: a container-shaped anchor (its own class +// implements Collection/Map - the LEAK_BUFFER wrapper shape) admitted at a +// LATE index position must leap the queue of ~28k other-tier anchors (the +// round-13 hotdog measurement: admission-order selection put the leak +// holder at position ~12-21k against ~4k walk coverage per search - +// deterministically unreachable). Also the tier-0 guarantee: a leak-tagged +// anchor leads the walk order unconditionally. +TEST_F(ReferenceChainsBfsTest, ContainerAnchorLeapsQueueAcrossLargeIndex) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr int kAnchorCount = 28000; + for (int i = 0; i < kAnchorCount; i++) { + jlong tag = 100 + i; + jlong class_tag = 500000 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, class_tag, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + // Containers only at late positions 24000..24003 - buried behind + // 24k other-tier anchors under any admission-order cursor. + bool container = (i >= 24000 && i <= 24003); + ReferenceChainsTestAccessor::primeClassShapeForTest(class_tag, + container); + } + // One leak-tagged anchor at the very tail - tier 0, must lead. + jlong leak_anchor = 100 + kAnchorCount; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, leak_anchor, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + frontier->setLeakTag(leak_anchor, 777); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + leak_anchor, 599999, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ReferenceChainsTestAccessor::primeClassShapeForTest( + 599999, false /* its tier comes from leak_tag, not shape */); + + std::vector selected = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, selected.size()) + << "28k+5 eligible anchors at a 16 budget must fill the selection"; + EXPECT_EQ(leak_anchor, selected[0]) + << "the leak-tagged anchor (tier 0) must lead the walk order"; + // The four container anchors are selected in this FIRST call despite + // positions 24000+ (tags 24100-24103). + for (jlong t : {24100, 24101, 24102, 24103}) { + EXPECT_NE(std::find(selected.begin(), selected.end(), t), selected.end()) + << "container anchor " << t + << " did not leap the other-tier queue"; + } + // Sanity: an early other-tier anchor also made the cut (cursor-fair + // fill from position 0). + EXPECT_NE(std::find(selected.begin(), selected.end(), (jlong)100), + selected.end()); + + tracker->stop(); +} + +// Round-15 fresh-admission priority: the hotdog wrapper is admitted +// LATE in a search (its holder class sits at sweep index 24627 of 33270, +// so admission lands at the anchor-index TAIL) - behind the whole fair +// container backlog measured at 1633 entries against 44-75-pass search +// lifetimes (the fair container cursor would reach it at pass ~102+, +// after the search is dead). A container admitted since the last +// collector call must be walked in the very NEXT call, ahead of the fair +// backlog; an unclassified fresh anchor rides the same lane (the wrapper +// admits one pass before reconcileAnchorClassShapes can classify its +// class), while a fresh NON-container waits in the other tier as before. +TEST_F(ReferenceChainsBfsTest, FreshContainerWalkedBeforeFairBacklog) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // A fair container backlog, "admitted long ago": 200 containers at + // positions 0..199. + constexpr int kBacklog = 200; + for (int i = 0; i < kBacklog; i++) { + jlong tag = 500 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, 700000 + i, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ReferenceChainsTestAccessor::primeClassShapeForTest(700000 + i, true); + } + // First collector call: all 200 anchors are fresh (nothing has had + // a first look yet), the lane keeps 16 and the rest spend their + // first look (they fall back to the fair container tier). + std::vector first = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, first.size()) << "200 eligible containers at a 16 budget"; + + // The late admission wave: a wrapper-class container at the index + // tail, a fresh classified NON-container (a fresh String static - it + // must NOT ride the fresh lane), and a fresh unclassified anchor (a + // genuinely new class - it must ride the lane, so the classification + // lag cannot lose the wrapper's fresh window). + const jlong wrapper_tag = 500 + kBacklog; + const jlong fresh_string_tag = 500 + kBacklog + 1; + const jlong fresh_unknown_tag = 500 + kBacklog + 2; + for (auto [tag, class_tag, container, prime] : + {std::make_tuple(wrapper_tag, (jlong)799001, true, true), + std::make_tuple(fresh_string_tag, (jlong)799002, false, true), + std::make_tuple(fresh_unknown_tag, (jlong)799003, false, + false /* deliberately unclassified */)}) { + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, class_tag, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + if (prime) { + ReferenceChainsTestAccessor::primeClassShapeForTest(class_tag, + container); + } + } + + std::vector second = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, second.size()); + EXPECT_EQ(wrapper_tag, second[0]) + << "the freshly admitted container must lead the walk order, not " + "wait behind the 184-container fair backlog"; + EXPECT_NE(std::find(second.begin(), second.end(), fresh_unknown_tag), + second.end()) + << "an unclassified fresh anchor must ride the fresh lane (the " + "wrapper admits a pass before reconcile classifies its class)"; + EXPECT_EQ(std::find(second.begin(), second.end(), fresh_string_tag), + second.end()) + << "a fresh NON-container stays in the other tier - the fresh lane " + "is the container lane, or fresh Strings would flood it"; + // The fair container lap still gets the leftover budget. Call 1's + // fresh picks (500-515) spent the whole budget, so the fair cursor + // never advanced - call 2's fair picks start at tag 500 again (a + // benign one-call overlap: walks are idempotent, and it only + // happens when fresh and fair coincide at a lap boundary). + EXPECT_NE(std::find(second.begin(), second.end(), (jlong)500), + second.end()) + << "the fair container lap must still advance with the leftover " + "budget"; + EXPECT_EQ(14, (int)std::count_if(second.begin(), second.end(), + [](jlong t) { + return t >= 500 && t < 500 + kBacklog; + })) + << "16 budget - 2 fresh picks = 14 fair-container picks"; + + // And the dropped fresh entries from call 1 (516-699 spent their + // first look) are still covered by the fair tier: a third call with + // no new admits must keep advancing the fair container cursor from + // wherever call 2 left it, not re-drain anything. + std::vector third = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest( + 16); + ASSERT_EQ(16u, third.size()); + EXPECT_EQ(std::find(third.begin(), third.end(), wrapper_tag), + third.end()) + << "the wrapper already had its first look - it must not be " + "re-selected while the fair cursor has 184 uncovered peers"; + + tracker->stop(); +} + +// Round-14 tier fairness: the other tier (everything not leak-tagged, not +// container-shaped) still reaches full coverage across wraps - a 40-anchor +// tier at a 16 budget covers all 40 in exactly 3 calls with no duplicates +// within a call. Containers must not permanently starve the rest of the +// index. +TEST_F(ReferenceChainsBfsTest, AnchorOtherTierFairCoverageAcrossWraps) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr int kOtherCount = 40; + for (int i = 0; i < kOtherCount; i++) { + jlong tag = 200 + i; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + tag, 300000 + i, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + ReferenceChainsTestAccessor::primeClassShapeForTest(300000 + i, false); + } + + std::vector all_selected; + for (int call = 0; call < 3; call++) { + std::vector selected = + ReferenceChainsTestAccessor:: + collectStaticFieldAnchorsForRotationForTest(16); + int expected = (call == 2) ? 8 : 16; + ASSERT_EQ(expected, (int)selected.size()) + << "call " << call << " should select " << expected; + std::set dedup(selected.begin(), selected.end()); + ASSERT_EQ(selected.size(), dedup.size()) + << "no anchor may be selected twice within a call"; + all_selected.insert(all_selected.end(), selected.begin(), + selected.end()); + } + std::set covered(all_selected.begin(), all_selected.end()); + ASSERT_EQ(40u, covered.size()) << "full other-tier coverage expected"; + for (int i = 0; i < kOtherCount; i++) { + EXPECT_NE(covered.find(200 + i), covered.end()) + << "other-tier anchor " << 200 + i << " never selected"; + } + + tracker->stop(); +} + +// Round-13/14 restart hygiene: discovered-instance tags are FRONTIER tags; +// restartSearch() resets the frontier table and _next_tag=1, so any +// surviving discovered slot either fails reconstructChain (observed on-pod: +// 'buildChainEvent failed ... reconstructChain failed for target_tag=8851') +// or, worse, resolves into a live new-search entry and emits a chain event +// for the WRONG OBJECT (the likely origin of the earlier noise-[B event). +// restartSearch() must clear them, along with the anchor index and its +// parallel arrays/cursors. +TEST_F(ReferenceChainsBfsTest, RestartSearchClearsDiscoveredInstanceTags) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // A watched candidate with one discovered instance recorded. + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 7); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(7, 4242, + false); + ASSERT_EQ(4242, ReferenceChainsTestAccessor::discoveredTagForTest(0, 0)); + ASSERT_EQ(1, ReferenceChainsTestAccessor::discoveredCountForTest(0)); + + // An anchor index entry (tag + parallel class tag) to confirm the + // index reset covers the parallel structures too. + ReferenceChainsTestAccessor::addToStaticAnchorIndexForTest( + 4242, 555, JVMTI_HEAP_REFERENCE_STATIC_FIELD); + (void)frontier; + + ReferenceChainsTestAccessor::restartSearchForTest(); + + EXPECT_EQ(0, ReferenceChainsTestAccessor::discoveredTagForTest(0, 0)) + << "stale discovered frontier tag survived restartSearch()"; + EXPECT_EQ(0, ReferenceChainsTestAccessor::discoveredCountForTest(0)); + EXPECT_TRUE(ReferenceChainsTestAccessor::anchorIndexIsEmptyForTest()) + << "anchor index (or its parallel arrays) survived restartSearch()"; + + tracker->stop(); +} + +// Round-19 (pod 289f8 — three JVMs of "canary search, 0/1 candidates found" +// while leak-tagged instances WERE intercepted and chains re-emitted): the +// marker->leak-tag design migration never updated the found criterion. +// _candidate_found_bits was set ONLY in heapReferenceCallback()'s marker-tag +// block, and markers are never set under leak tags ("no marker tags — using +// leak tags now"), so the chase was structurally unresolvable and every +// search ended via frontier-cap/no-progress instead of "all candidates +// found". The leak-tag-world criterion: a leak-tag-target chain +// (target_tag >= LEAK_TAG_BASE) built for a candidate slot marks that slot +// found and records its canary link; a noise chain (no leak tag) must NOT. +TEST_F(ReferenceChainsBfsTest, LeakTagChainMarksCanaryFound) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // Slot 0: candidate klass 7 whose discovered instance carries a leak tag + // (entry.leak_tag set — the shape a walk + tagLeakInstances correlation + // produces; target_tag becomes the leak tag). Slot 1: candidate klass 9 + // with a plain noise discovered instance, no leak tag. + ReferenceChainsTestAccessor::setCandidateCountForTest(2); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, 7); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(1, 9); + // depth=1 at a durable STATIC_FIELD root (root_kind 8) so the retention + // filter (suppressChainEvent: depth==0, or depth==1 at a transient + // root) keeps both chains. + ASSERT_TRUE(frontier->insert(4242, 0, 7, 1, FrontierEntryState::FRONTIER, + 8, /*class_tag=*/0, -1, /*kind=*/0)); + frontier->setLeakTag(4242, 1073742079); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(7, 4242, true); + ASSERT_EQ(4242, ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + ASSERT_TRUE(frontier->insert(5353, 0, 9, 1, FrontierEntryState::FRONTIER, + 8, 0, -1, 0)); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(9, 5353, false); + ASSERT_EQ(5353, ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(1, 0)); + + ReferenceChainsTestAccessor::buildDiscoveredInstanceChainsForTest(7, 1); + ReferenceChainsTestAccessor::buildDiscoveredInstanceChainsForTest(9, 1); + + // Slot 0 found via its leak-tag chain (bit 0 set, link recorded); + // slot 1's noise chain did not mark anything found. + EXPECT_EQ(1ULL, ReferenceChainsTestAccessor::candidateFoundBitsForTest()); + EXPECT_EQ(4242, ReferenceChainsTestAccessor::candidateFrontierTagForTest(0)); + EXPECT_EQ(0, ReferenceChainsTestAccessor::candidateFrontierTagForTest(1)) + << "noise-target chain must not mark the canary found"; + + tracker->stop(); +} + +// B' push site 1 - DEMOTION TIME (find-anchor-holder-eviction / _static_anchor_fifo): +// when improveChain() replaces a root-attached durable (STATIC_FIELD/ +// JNI_GLOBAL) entry with a deeper chain-attached path, the entry is leaving +// the anchor tier's eligible population at exactly that moment - the push +// must fire right there. This matters because the sweep gate only re-laps +// while the class count is in flux, so a post-lap demotion would otherwise +// never see another static edge onto the entry. Driven through the REAL +// expandFrontier batch walk (mock FollowReferences + real heapReferenceCallback). +TEST_F(ReferenceChainsBfsTest, DemotionPushFiresWhenImproveChainEvictsRootAttachedStatic) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int parentNode = addNode(); + int holderNode = addNode(); + + // The chain edge whose delivery demotes the holder. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, parentNode, holderNode, -1}, + }; + + // Seed exactly the pre-demotion shape: holder root-attached STATIC + // (anchor-eligible), parent a root-attached frontier object whose + // expansion delivers the deeper chain edge. node_tags make both + // resolvable by mock_GetObjectsWithTags for the batch walk. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + node_tags[parentNode] = 104; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(104); + + int edges = 0; + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, + &edges); + // The chain edge was delivered: the holder's entry is now chain-attached + // (improveChain replaced the depth-0 root-attached admission), and the + // demotion pushed its tag into the at-risk FIFO. + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(105, &entry)); + EXPECT_EQ(104, entry.parent_tag); + EXPECT_EQ(1u, entry.depth); + ASSERT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); + + // Re-walking the same edge must NOT push twice: improveChain refuses + // (new depth 1 is not > current 1), and the set dedupes regardless. + int edges2 = 0; + ReferenceChainsTestAccessor::pushPendingExpandForTest(104); + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, + &edges2); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + + tracker->stop(); +} + +// B' push site 2 - SWEEP TIME: the static sweep's class->field edge onto an +// already-admitted CHAIN-ATTACHED entry (the admission-order eviction shape: +// born as a non-root child, never root-attached at all) must feed the FIFO. +// Driven through a full runPass so the real admitStaticFieldRoots sweep (and +// its FollowReferences) delivers the edge, and the rotation phase of the +// SAME pass drains the FIFO - the push counter survives the drain. +TEST_F(ReferenceChainsBfsTest, SweepPushFiresOnStaticEdgeOntoChainAttachedHolder) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=100")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int classNode = addNode(); + int holderNode = addNode(); + // A child reachable ONLY from the holder: nothing in the scripted + // graph walks holderNode except the FIFO-drained anchor walk, so the + // child's admission after runPass is the end-to-end evidence that the + // sweep pushed the holder, the rotation phase drained it, and + // walkStaticFieldAnchors walked it. + int holderChildNode = addNode(); + addClass((void *)&node_tags[classNode], "Lcom/rc/statics/ChainBornHolder;"); + + script = { + // Only the sweep's static edge onto the holder, plus the holder's + // own child edge for the anchor walk to admit: nothing else reaches + // holderNode or holderChildNode, so the only possible push is the + // static-edge-onto-chain-attached site. + {JVMTI_HEAP_REFERENCE_STATIC_FIELD, classNode, holderNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, holderChildNode, -1}, + }; + + // Seed the born-chain-attached shape the eviction leaves: holder already + // a non-root child (parent 104, depth 1). The sweep's static edge below + // hits maybeUpgradeRootAttachedRootKind's documented parent_tag != 0 + // refusal and must fall into the B' push instead. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + // The child is untagged (0): the anchor walk's admission assigns it a + // fresh frontier tag, observable via node_tags after the pass. + ASSERT_EQ(0, node_tags[holderChildNode]); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + bool truncated = true; + ASSERT_TRUE(tracker->runPass(&mock_jvmti, &mock_jni, &truncated)); + + // The rotation phase of the same pass drained the pushed tag into the + // anchor walk, and the walk admitted the holder's child - the push + // itself left no residue in the FIFO (drained empty) and never + // re-attributed the holder's entry (re-rooting is the documented + // refusal that motivated the FIFO in the first place). The child's + // admission is observed via tags_ever_assigned rather than node_tags: + // the frontier drained empty and the search COMPLETED in this same + // pass, so releaseSearchTags() has already cleared every live JVMTI + // tag (including the child's and the holder's) by the time runPass + // returns - the frontier table's own records survive that, the tag + // map does not. + EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + jlong childTag = tags_ever_assigned[holderChildNode]; + ASSERT_GT(childTag, 0) << "the FIFO-drained anchor walk never admitted " + "the holder's child"; + FrontierEntry childEntry{}; + ASSERT_TRUE(frontier->lookup(childTag, &childEntry)); + EXPECT_EQ(105, childEntry.parent_tag); + EXPECT_EQ(2u, childEntry.depth); + // The holder's entry is untouched by the push - the B' feed records the + // at-risk shape, it never re-attributes the entry. + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(105, &entry)); + EXPECT_EQ(104, entry.parent_tag); + EXPECT_EQ(1u, entry.depth); + + tracker->stop(); +} + +// B' mechanics: a chain-attached holder drained from the at-risk FIFO is +// descend-walked and intercepts a leak chunk 3 hops below it - the repair +// for the population the root-attached collector demonstrably cannot +// select (the negative control below). Mirrors +// StaticAnchorRotationWalksRootAttachedStaticHolders's walk-phase shape, +// but with the holder chain-attached and FIFO-sourced. +TEST_F(ReferenceChainsBfsTest, AtRiskAnchorFifoDrainAndWalkIntercept) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x6001, *chunkCls = (void *)0x6002; + addClass(holderCls, "Lcom/rc/descendwalk/ChainAttachedHolder;"); + int chunk = addClass(chunkCls, "Lcom/rc/descendwalk/ChainChunk;"); + + int parentNode = addNode(); + int holderNode = addNode(); + int tableNode = addNode(); + int entryNode = addNode(); + int leakChunk = addNode(); + const jlong leak_tag = ReferenceChainsTestAccessor::leakTagBase(); + node_tags[leakChunk] = leak_tag; + + // Chain: parent -> holder -> table -> entry -> leak chunk. Only the + // holder's own subtree is scripted for the walk below (the direct + // walkStaticAnchors drive never runs the roots/expand phases). + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, tableNode, -1}, + {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, + {JVMTI_HEAP_REFERENCE_FIELD, entryNode, leakChunk, chunk}, + }; + + // Seed the holder exactly as the eviction leaves it: chain-attached + // (parent_tag = 104, depth 1, no root_kind). node_tags[holderNode] = 105 + // makes the tag resolvable by mock_GetObjectsWithTags, exactly like the + // root-attached test's own 101 mapping. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + // root_kind = 0: the entry is chain-attached, and a non-root entry's + // edge kind is not recorded (FrontierEntry::root_kind's own comment). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); + // The chain's root: TRANSIENT (stack local), so the collector's durable + // root-kind filter skips it too - the whole table is un-selectable. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + // Negative control: the root-attached collector selects NOTHING from a + // table holding only a chain-attached holder and a transient root - + // the pre-B' behavior that stranded the dual-reachable population. + std::vector selected = + ReferenceChainsTestAccessor::collectStaticFieldAnchorsForRotationForTest(4); + ASSERT_TRUE(selected.empty()); + + // B': push + drain + walk reaches what the collector cannot. + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(105, 28366); + std::vector drained; + // 16 = ReferenceChainTracker::STATIC_ANCHOR_FIFO_DRAIN (private), the + // same per-pass drain cap runPassManualWalk() uses. + ASSERT_EQ(1, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, drained)); + ASSERT_EQ(1u, drained.size()); + EXPECT_EQ(105, drained[0].tag); + EXPECT_EQ(28366u, drained[0].klass_id); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + + int edges = 0; + std::vector drained_tags; + for (const auto &at_risk : drained) { + drained_tags.push_back(at_risk.tag); + } + ReferenceChainsTestAccessor::walkStaticAnchorFifoForTest( + &mock_jvmti, &mock_jni, drained_tags, 1000, &edges, nullptr); + jlong table_ftag = tags_ever_assigned[tableNode]; + jlong entry_ftag = tags_ever_assigned[entryNode]; + jlong chunk_ftag = tags_ever_assigned[leakChunk]; + ASSERT_GT(table_ftag, 0) << "table array was not reached by the anchor walk"; + ASSERT_GT(entry_ftag, 0) << "Entry was not reached one hop below table"; + ASSERT_NE(chunk_ftag, leak_tag) + << "leak-tagged chunk inside the chain-attached holder was never " + "intercepted"; + EXPECT_EQ(leak_tag, + ReferenceChainsTestAccessor::frontierLeakTag(chunk_ftag)); + FrontierEntry chunk_entry{}; + ASSERT_TRUE(frontier->lookup(chunk_ftag, &chunk_entry)); + EXPECT_EQ(entry_ftag, chunk_entry.parent_tag); + EXPECT_EQ(4u, chunk_entry.depth); + + tracker->stop(); +} + +// B' requeue mechanics: a truncated anchor walk reports exactly the +// RESOLVED-but-unwalked tags, and requeueStaticAnchorFifoFront() restores +// them to the FIFO front in order with a consistent membership set - so +// an at-risk holder that lost its budget turn keeps it for the next pass +// instead of waiting for the next sweep lap. +TEST_F(ReferenceChainsBfsTest, TruncatedAnchorWalkRequeuesUnwalkedFifoTags) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x7001; + addClass(holderCls, "Lcom/rc/descendwalk/RequeueHolder;"); + + int holderANode = addNode(); + int holderBNode = addNode(); + int tableNode = addNode(); + int entryNode = addNode(); + + // Holder A's subtree is deep enough that a budget of 2 truncates the + // walk after A (two edges admitted, budget exhausted on the descend); + // holder B then must come back unwalked. B has no scripted subtree - + // its walk would be a no-op anyway, which is exactly what makes the + // unwalked report attributable to the truncation, not to content. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderANode, tableNode, -1}, + {JVMTI_HEAP_REFERENCE_ARRAY_ELEMENT, tableNode, entryNode, -1}, + }; + + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderANode] = 105; + node_tags[holderBNode] = 106; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 104, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 106, 104, 1, FrontierEntryState::FRONTIER, 0)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 104, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(105, 2001); + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(106, 2002); + std::vector drained; + // 16 = STATIC_ANCHOR_FIFO_DRAIN (private), the per-pass drain cap. + ASSERT_EQ(2, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, drained)); + ASSERT_EQ(2u, drained.size()); + EXPECT_EQ(105, drained[0].tag); + EXPECT_EQ(106, drained[1].tag); + + int edges = 0; + std::vector drained_tags; + for (const auto &at_risk : drained) { + drained_tags.push_back(at_risk.tag); + } + std::vector unwalked; + ReferenceChainsTestAccessor::walkStaticAnchorFifoForTest( + &mock_jvmti, &mock_jni, drained_tags, 2, &edges, &unwalked); + ASSERT_EQ(1u, unwalked.size()); + EXPECT_EQ(106, unwalked[0]); + EXPECT_NE(0, tags_ever_assigned[tableNode]) + << "holder A's walk never ran - the truncation happened too early"; + + // Requeue exactly what the caller-side filter in runPassManualWalk() + // would requeue (here: everything unwalked, both FIFO-sourced), keeping + // the drained entries' klass so the per-class occupancy is restored. + std::vector requeue; + for (jlong unwalked_tag : unwalked) { + for (const auto &at_risk : drained) { + if (unwalked_tag == at_risk.tag) { + requeue.push_back(at_risk); + break; + } + } + } + ReferenceChainsTestAccessor::requeueStaticAnchorFifoFrontForTest(requeue); + EXPECT_EQ(1u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(106)); + EXPECT_FALSE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(105)); + std::vector redrained; + ASSERT_EQ(1, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, redrained)); + ASSERT_EQ(1u, redrained.size()); + EXPECT_EQ(106, redrained[0].tag); + + tracker->stop(); +} + +// Round 16, fix A: a this-field self-edge is REAL in the heap - every +// java.util.Collections$Synchronized* holder carries mutex == this - so +// walking such a holder's own subtree (rotation anchor walk or BFS descent) +// re-reports the holder as its own child through that field. On the pod +// (round-15 evidence, ev-leaktag-onpod-round15-results) that edge's +// improveChain() "improvement" (depth = holder.depth + 1 > 0) overwrote the +// LEAK_BUFFER wrapper's root-attached entry with a parent==its-own-tag +// chain-attached one (probe: parent==tag root_kind=0 referrer_klass=) - collector-invisible ever after, because the entry left the +// parent_tag == 0 population just as B' got cap-pinned by floods. The guard +// refuses the self-parent (and counts it), so the entry keeps its +// root-attached shape and stays collector-selectable. Driven through the +// real expandFrontier batch walk, the same harness as the demotion-push +// test above it - which also keeps proving the LEGIT deeper-path demotion +// still works (that test must keep passing). +TEST_F(ReferenceChainsBfsTest, + SelfEdgeFieldDoesNotDemoteRootAttachedStaticHolder) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + void *holderCls = (void *)0x7101; + int holderClsIdx = + addClass(holderCls, "Ljava/util/Collections$SynchronizedRandomAccessList;"); + int holderNode = addNode(); + + // The holder's own self-edge: referrer == referee == holder (the + // mutex == this field), exactly as its rotation walk re-reports it. + script = { + {JVMTI_HEAP_REFERENCE_FIELD, holderNode, holderNode, holderClsIdx}, + }; + + // Seed the pre-demotion shape: holder root-attached STATIC_FIELD + // (anchor-eligible), admitted at depth 0. The batch walk from its own + // pending-expand slot delivers the self-edge with the holder as BOTH + // referrer and referee. + FrontierTable *frontier = tracker->frontierTable(); + node_tags[holderNode] = 105; + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 105, 0, 0, FrontierEntryState::FRONTIER, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::pushPendingExpandForTest(105); + + // A delivered self-edge trips BOTH sibling guards: improveChain refuses, + // and the already-admitted block's else-if then offers the same + // self-parent to reparentToDurableRoot, which refuses too. + int edges = 0; + ReferenceChainsTestAccessor::expandFrontierForTest(&mock_jvmti, &mock_jni, + &edges); + + // The self-edge was delivered and refused: the entry keeps its + // root-attached shape (the collector's parent_tag == 0 eligibility) + // and no demotion push fired. + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(105, &entry)); + EXPECT_EQ(0, entry.parent_tag); + EXPECT_EQ((u8)JVMTI_HEAP_REFERENCE_STATIC_FIELD, entry.root_kind); + EXPECT_EQ(0u, entry.depth); + EXPECT_EQ(0u, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + + // The sibling guards and the direct table calls agree: a self-parent is + // refused by both improvement paths, unconditionally. + EXPECT_FALSE(frontier->improveChain(105, 105, 0, 5, 0, -1, 0, 0)); + EXPECT_FALSE(frontier->reparentToDurableRoot(105, 105, 0, -1, 0)); + + tracker->stop(); +} + +// Round 16, fix B (pod round-15 measurement, ev-leaktag-onpod-round15- +// results): the B' at-risk FIFO sat cap-pinned at 1024 because three +// classes flooded it (klass 1: 1396 pushes, klass 215: 1063, klass 1733: +// 988+), so the LEAK_BUFFER wrapper's pushes (klass 28366) were dropped at +// the cap check and the lane designed to repair exactly its demotion never +// carried it. The per-class quota keeps one class from owning the lane: at +// 64 per class, a full 1024-entry FIFO necessarily holds >= 16 distinct +// classes, the flood's excess is dropped and counted, and the wrapper's +// fresh-class push lands. Occupancy is maintained exactly across push / +// drain / requeue, and a saturated-but-diverse FIFO drops newcomers at the +// cap rather than evicting anyone. +TEST_F(ReferenceChainsBfsTest, AtRiskFifoPerClassQuotaKeepsFloodOut) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + const u32 quota = + ReferenceChainsTestAccessor::kAtRiskPerKlassCap; + ASSERT_EQ(64u, quota); + + // The flood: only the first `quota` pushes of one class land. + for (int i = 0; i < 70; i++) { + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(1000 + i, + 1733); + } + EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + // The flood's first `quota` tags hold their slots and the excess is + // dropped at the quota check - absent from the FIFO, not queued. + EXPECT_TRUE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest( + 1000 + (int)quota - 1)); + EXPECT_FALSE(ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest( + 1000 + (int)quota)); + + // The wrapper's push (a different class) lands despite the flood - + // exactly the push the pod dropped. + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE( + ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(2000)); + + // Tag dedupe is unchanged: the same tag never enters twice. + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(2000, 28366); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + + // Full drain: order preserved (flood first, newcomer last), occupancy + // erased with the entries - the next flood can land again (it never + // owns MORE than its quota, but it is not permanently locked out + // either). + std::vector drained; + ASSERT_EQ((int)quota + 1, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 1024, drained)); + ASSERT_EQ(quota + 1, drained.size()); + EXPECT_EQ(1000, drained.front().tag); + EXPECT_EQ(1733u, drained.front().klass_id); + EXPECT_EQ(2000, drained.back().tag); + EXPECT_EQ(28366u, drained.back().klass_id); + for (int i = 0; i < 70; i++) { + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(1000 + i, + 1733); + } + EXPECT_EQ(quota, ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + + // Partial drain at the real per-pass rate (16/pass): the flood's + // occupancy is 64 - 16 = 48 after the drain, so its next push lands + // (refilling its share as it drains - the flood self-throttles, it + // never starves the lane). + std::vector partial; + ASSERT_EQ(16, ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, partial)); + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(3000, 1733); + EXPECT_EQ(quota - 16 + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_TRUE( + ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(3000)); + + // Requeue restores occupancy exactly: the requeued entries occupy + // their slots again and drain in FIFO order. Note this may put a class + // momentarily ABOVE its quota - the quota gates NEW pushes, it does + // not evict truncated-walk requeues of entries that legitimately + // held their slots before the drain. + std::vector requeue(partial.begin(), + partial.end()); + ReferenceChainsTestAccessor::requeueStaticAnchorFifoFrontForTest(requeue); + EXPECT_EQ(quota + 1, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + std::vector after_requeue; + ASSERT_EQ(16, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 16, after_requeue)); + EXPECT_EQ(1000, after_requeue.front().tag); + + // Saturated-but-diverse: 16 distinct classes at exactly their quota + // fill the 1024 cap, and the newcomer is dropped at the CAP (legitimate + // saturation - no eviction), not because of any flood. The remaining + // count after the drains above: 64 flood entries - 16 (partial) - 16 + // (post-requeue) + 1 (tag 3000) = quota - 16 + 1. + std::vector rest; + ASSERT_EQ((int)quota - 16 + 1, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 1024, rest)); + for (u32 klass = 1; klass <= 16; klass++) { + for (u32 i = 0; i < quota; i++) { + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest( + 10000 + klass * 100 + i, 4000 + klass); + } + } + EXPECT_EQ(1024u, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + // The 16 saturated classes sit exactly at their per-class quota (no + // quota drop is even possible at exactly `quota` pushes), so the + // newcomer's absence below is the CAP's doing, not the quota's. + ReferenceChainsTestAccessor::pushStaticAnchorFifoForTest(20000, 28366); + EXPECT_EQ(1024u, + ReferenceChainsTestAccessor::staticAnchorFifoSizeForTest()); + EXPECT_FALSE( + ReferenceChainsTestAccessor::staticAnchorFifoContainsForTest(20000)); + + // Leave the FIFO drained: this test is the one that saturates it, and + // the reset() seam above now clears it for the next test regardless - + // but a drained ending also keeps this test order-independent even if + // that seam ever regresses again. + std::vector final_drain; + ASSERT_EQ(1024, + ReferenceChainsTestAccessor::drainStaticAnchorFifoForTest( + 1024, final_drain)); + + tracker->stop(); +} + +// Retention-edge labels (fillHopEdgeLabels()/hopLabelClassFor()): the +// emitted chain's per-hop labels must decode the JVMTI-SPECIFICATION field +// ordinal captured at admission - the interface offset, the superclass-chain +// order, the interface-referrer branch - and degrade to the edge KIND on any +// undecodable hop, never a fabricated name (the fail-safe contract: a wrong +// numbering on an unverified JVM degrades, it does not lie). +TEST_F(ReferenceChainsBfsTest, HopEdgeLabelsDecodeSpecFieldOrdinals) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=1000")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + // Hierarchy mirroring the spec's own numbering example shape: + // interface IBase { int p; } 1 own field + // interface ISub extends IBase { int x; } 1 own field + // class Base { int base_f; } 1 own field, no super + // class Holder extends Base implements ISub + // { int holder_a; Object leakList; } + // interface ISink { Object CONST_A; Object CONST_B; } + // Spec ordinal spaces (jvmtiHeapReferenceInfoField): + // Holder (class branch): base = ISub(1) + IBase(1) = 2 (transitive + // interfaces, each once); then the superclass chain root-first: + // base_f@2; then own fields in GetClassFields order: + // holder_a@3, leakList@4. + // ISink (interface branch): base = superinterfaces' fields = 0; own + // fields in GetClassFields order: CONST_A@0, CONST_B@1. + void *ibase = (void *)0x5001, *isub = (void *)0x5002, *base = (void *)0x5003, + *holder = (void *)0x5004, *isink = (void *)0x5005; + addClass(ibase, "Lcom/rc/labels/IBase;"); + addClass(isub, "Lcom/rc/labels/ISub;"); + addClass(base, "Lcom/rc/labels/Base;"); + addClass(holder, "Lcom/rc/labels/Holder;"); + addClass(isink, "Lcom/rc/labels/ISink;"); + ReferenceChainsTestAccessor::resolveLoadedClasses(&mock_jvmti, &mock_jni); + // resolveLoadedClasses minted each class's raw (negative) tag via the + // mock's tag map - read them back for the decoder's tag->class lookup + // and the entries' referrer_class_tag values. + auto tagOf = [&](void *k) -> jlong { return tags[k]; }; + field_decode_classes[tagOf(holder)] = holder; + field_decode_classes[tagOf(isink)] = isink; + field_decode_classes[tagOf(base)] = base; + // ibase deliberately NOT registered into field_decode_classes: an + // unresolvable referrer class below must degrade to a kind label. + field_decode_hierarchy[ibase] = {true, nullptr, {}, {{(void *)0x6001, "p"}}}; + field_decode_hierarchy[isub] = + {true, nullptr, {ibase}, {{(void *)0x6002, "x"}}}; + field_decode_hierarchy[base] = {false, nullptr, {}, {{(void *)0x6003, "base_f"}}}; + field_decode_hierarchy[holder] = + {false, base, {isub}, + {{(void *)0x6004, "holder_a"}, {(void *)0x6005, "leakList"}}}; + field_decode_hierarchy[isink] = + {true, nullptr, {}, + {{(void *)0x6006, "CONST_A"}, {(void *)0x6007, "CONST_B"}}}; + + FrontierTable *frontier = tracker->frontierTable(); + // Chain: [chunk(3)] <- Base.base_f(ordinal 0 over Base's space) <- + // [value2(2), class Base] <- Holder.leakList(ordinal 4, the static root + // edge with the declaring class as referrer) <- [static value(1), class + // Holder] <- [class Holder (the ROOT TYPE - buildChainEvent() appends + // the root-attached entry's referrer_class_tag for static-field roots)]. + // Interior hops decode against the PARENT entry's class_tag. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EDGE, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, + /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), + /*referrer_field_index=*/4, /*edge_kind=*/0, + /*referrer_class_tag=*/tagOf(holder))); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, 1, 1, FrontierEntryState::EDGE, /*root_kind=*/0, + /*referrer_klass=*/0, /*class_tag=*/tagOf(base), + /*referrer_field_index=*/3, JVMTI_HEAP_REFERENCE_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 3, 2, 2, FrontierEntryState::EDGE, /*root_kind=*/0, + /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), + /*referrer_field_index=*/0, JVMTI_HEAP_REFERENCE_FIELD)); + + ReferenceChainEvent event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, /*target_tag=*/3, &event)); + ASSERT_EQ(4u, event._hops.size()); + // Leaf first: chunk is retained via Base.base_f (parent entry's class is + // Base, ordinal 0 in Base's own space), then value2 via Holder's + // holder_a (ordinal 3 = interface offset 2 + Base's 1 + own position 0), + // then the static root edge's field name leakList (ordinal 4), then the + // root-type hop (class Holder) - the root edge itself, kind label only. + EXPECT_EQ("base_f", event._hops[0].edge_label); + EXPECT_EQ("holder_a", event._hops[1].edge_label); + EXPECT_EQ("leakList", event._hops[2].edge_label); + EXPECT_EQ("static_field", event._hops[3].edge_label); + int expectedRootType = Profiler::instance()->lookupClass( + "com/rc/labels/Holder", strlen("com/rc/labels/Holder")); + ASSERT_NE(-1, expectedRootType); + EXPECT_EQ((u32)expectedRootType, event._hops[3].klass_id); + + // Interface-referrer branch: ISink's own-field ordinals have NO + // superclass-chain component (base = superinterfaces' fields only). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 4, 0, 0, FrontierEntryState::EDGE, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, + /*referrer_klass=*/0, /*class_tag=*/tagOf(isink), + /*referrer_field_index=*/1, /*edge_kind=*/0, + /*referrer_class_tag=*/tagOf(isink))); + ReferenceChainEvent iface_event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, /*target_tag=*/4, &iface_event)); + // Same root-type append as above: the root-side end gains ISink (the + // declaring class of the static field) plus its kind-only root edge. + ASSERT_EQ(2u, iface_event._hops.size()); + EXPECT_EQ("CONST_B", iface_event._hops[0].edge_label); + EXPECT_EQ("static_field", iface_event._hops[1].edge_label); + int expectedSinkRoot = Profiler::instance()->lookupClass( + "com/rc/labels/ISink", strlen("com/rc/labels/ISink")); + ASSERT_NE(-1, expectedSinkRoot); + EXPECT_EQ((u32)expectedSinkRoot, iface_event._hops[1].klass_id); + + // Fail-safe: a referrer class that cannot be resolved degrades to the + // edge KIND label, never a fabricated name. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EDGE, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, + /*referrer_klass=*/0, /*class_tag=*/tagOf(holder), + /*referrer_field_index=*/0, /*edge_kind=*/0, + /*referrer_class_tag=*/tagOf(ibase))); + ReferenceChainEvent degraded_event; + ASSERT_TRUE(ReferenceChainsTestAccessor::buildChainEventForTest( + &mock_jvmti, &mock_jni, /*target_tag=*/5, °raded_event)); + // Both hops degrade to kind labels: the holder hop's referrer class + // (IBase) is deliberately unregistered from the decoder, and the + // appended root-type hop (IBase itself) carries no field identity. + ASSERT_EQ(2u, degraded_event._hops.size()); + EXPECT_EQ("static_field", degraded_event._hops[0].edge_label); + EXPECT_EQ("static_field", degraded_event._hops[1].edge_label); + int expectedIbaseRoot = Profiler::instance()->lookupClass( + "com/rc/labels/IBase", strlen("com/rc/labels/IBase")); + ASSERT_NE(-1, expectedIbaseRoot); + EXPECT_EQ((u32)expectedIbaseRoot, degraded_event._hops[1].klass_id); + + tracker->stop(); +} + +// PRIORITY_EXPAND_CAP backpressure: with the fast lane at the cap, the +// rotation collectors must stop pushing. The uncapped queue is what starved +// the BFS on-pod (39k->103k entries while _pending_expand never drained a +// single batch - rotation inflow 256/pass exceeded the deadline-bounded +// drain ~150-300/pass on every pass). +TEST_F(ReferenceChainsBfsTest, PriorityExpandCapStopsRotationCollectorPushes) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // One eligible stale-EXPANDED entry. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + for (size_t i = 0; i < ReferenceChainsTestAccessor::priorityExpandCap(); i++) { + ReferenceChainsTestAccessor::pushPriorityExpand((jlong)(100 + i)); + } + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(10); + EXPECT_TRUE(selected.empty()) + << "collector must stop pushing once _priority_expand hits the cap"; + + // With the lane drained (a pass's expand phase consumed it), the + // collector selects again. + ReferenceChainsTestAccessor::clearPriorityExpand(); + selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(10); + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + + tracker->stop(); +} + +// The stale-expansion rotation must select leak parents from +// _leak_parent_fanout ahead of the blind table lap: the fanout entries are +// the EXPANDED parents that actually lead to watched leak-klass children, +// and neither the blind lap (~table_size/budget passes, hundreds live) nor +// the growth-gated leak-accumulation tier reaches them in steady state. +TEST_F(ReferenceChainsBfsTest, StaleRotationPrefersLeakParentsOverBlindLap) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kLeafKlass = 987; + ReferenceChainsTestAccessor::setWatchedLeakKlassIdsForTest({kLeafKlass}); + + // Fanout parent 1 and an unrelated stale-EXPANDED entry 3. Parent 1 + // needs a non-zero class_tag: trackLeakAccumulation() attributes via the + // parent entry's class_tag and skips entries without one. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD, /*referrer_klass=*/0, + /*class_tag=*/42)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 3, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ReferenceChainsTestAccessor::trackLeakAccumulation(frontier, kLeafKlass, 1, 10); + + // Budget 1: the fanout parent wins over the blind-lap entry. + std::vector selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(1); + ASSERT_EQ(1u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + + // Budget covering both: fanout parent first, blind lap fills the rest. + ReferenceChainsTestAccessor::clearPriorityExpand(); + selected = + ReferenceChainsTestAccessor::collectStaleExpandedEntriesForRotation(2); + ASSERT_EQ(2u, selected.size()); + EXPECT_EQ((jlong)1, selected[0]); + EXPECT_EQ((jlong)3, selected[1]); + + tracker->stop(); +} + +// reparentToDurableRoot: a depth-1 entry first admitted through a transient +// root (stack/JNI local) is re-parented to a durable root-attached parent +// at equal depth - the case improveChain() cannot express (it requires a +// strictly deeper path), and exactly the hotdog shape where the singleton +// collection is a depth-0 static root and its elements depth 1. +TEST_F(ReferenceChainsBfsTest, ReparentToDurableRootSwapsTransientForDurable) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + // tag 1: transient root (old parent). tag 2: target at depth 1 under it. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 1, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 2, 1, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // tag 5: durable static root (new parent). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 5, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + // tag 6: another transient root - must never be swapped TO. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 6, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + + EXPECT_TRUE(frontier->reparentToDurableRoot(2, 5, 42)); + FrontierEntry entry{}; + ASSERT_TRUE(frontier->lookup(2, &entry)); + EXPECT_EQ((jlong)5, entry.parent_tag); + EXPECT_EQ((u32)42, entry.referrer_klass); + + // Transient new parent: no swap (would trade one noise root for + // another). + EXPECT_FALSE(frontier->reparentToDurableRoot(2, 6, 43)); + ASSERT_TRUE(frontier->lookup(2, &entry)); + EXPECT_EQ((jlong)5, entry.parent_tag) << "parent must be unchanged"; + + // Depth-2 targets are out of scope (judging root durability there + // would require walking both chains). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 7, 2, 2, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + EXPECT_FALSE(frontier->reparentToDurableRoot(7, 5, 44)); + + tracker->stop(); +} + +// recordDiscoveredInstance eviction: noise instances fill discovery slots +// first-come-first-served, but a leak-correlated discovery must evict a +// noise slot when all are full - without eviction, the 8 noise instances +// observed on-pod permanently blocked every later leak-tagged instance of +// the watched class. +TEST_F(ReferenceChainsBfsTest, RecordDiscoveredInstanceEvictsNoiseSlots) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kKlass = 3; + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, kKlass); + + const int cap = ReferenceChainsTestAccessor::maxDiscoveredPerClass(); + for (int d = 0; d < cap; d++) { + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, /*tag=*/100 + d, /*leak_correlated=*/false); + } + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + // Noise beyond the cap is dropped, slots unchanged. + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, 108, false); + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + EXPECT_EQ((jlong)100, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + + // Leak-correlated discovery evicts the first noise slot (tag 100 has + // no frontier entry -> treated as uncorrelated). + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, 200, true); + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + EXPECT_EQ((jlong)200, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + EXPECT_EQ((jlong)101, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 1)); + + // Once every slot is leak-correlated, a further leak discovery is + // dropped (no eviction of real signal). + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 200, 0, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + frontier->setLeakTag(200, ReferenceChainsTestAccessor::leakTagBase() + 1); + // Entries for the remaining noise slots so the eviction scan finds all + // slots leak-tagged. + for (int d = 1; d < cap; d++) { + jlong tag = 101 + (d - 1); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, tag, 0, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + frontier->setLeakTag(tag, ReferenceChainsTestAccessor::leakTagBase() + 2); + } + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest( + kKlass, 201, true); + EXPECT_EQ(cap, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + for (int d = 0; d < cap; d++) { + EXPECT_NE((jlong)201, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, d)); + } + + tracker->stop(); +} + +// correlateAdmittedLeakTag: a tracked instance the BFS admitted BEFORE it +// was leak-tagged carries a frontier tag on the object; correlating stores +// the leak tag ON the entry (chain events then emit targetTag = leak tag) +// and records the instance as discovered. Never retags the object. +TEST_F(ReferenceChainsBfsTest, CorrelateAdmittedLeakTagSetsEntryAndDiscovers) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true:hops=64:budget=64")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + FrontierTable *frontier = tracker->frontierTable(); + + constexpr u32 kKlass = 3; + constexpr jlong kLeakTag = 0x40000000LL + 5; + ReferenceChainsTestAccessor::setCandidateCountForTest(1); + ReferenceChainsTestAccessor::setCandidateKlassIdForTest(0, kKlass); + + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 300, 0, 1, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + EXPECT_TRUE(tracker->correlateAdmittedLeakTag(300, kLeakTag, kKlass)); + EXPECT_EQ(kLeakTag, (jlong)ReferenceChainsTestAccessor::frontierLeakTag(300)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + EXPECT_EQ((jlong)300, + ReferenceChainsTestAccessor::candidateDiscoveredTagForTest(0, 0)); + + // Idempotent: an already-correlated entry just returns true. + EXPECT_TRUE(tracker->correlateAdmittedLeakTag(300, kLeakTag, kKlass)); + EXPECT_EQ(kLeakTag, (jlong)ReferenceChainsTestAccessor::frontierLeakTag(300)); + + // Unknown tag: no crash, no discovery side effects. + EXPECT_FALSE(tracker->correlateAdmittedLeakTag(999, kLeakTag, kKlass)); + EXPECT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + tracker->stop(); +} + +// Retention-explanation gate on the discovered-instance chains (depth==0 +// always suppressed; depth==1 suppressed only for TRANSIENT roots - a +// depth-1 chain from a durable root is the real direct-retention shape): +// transient depth-1 must NOT be cached, durable depth-1 and depth-2 must. +TEST_F(PollWatchedTargetsTest, DiscoveredChainGateSuppressesTransientDepthOne) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*klass_id=*/3, /*rep=*/(jweak)obj); + + // First poll populates the candidate slots from LivenessTracker's + // population. No discovered instances yet. + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + FrontierTable *frontier = tracker->frontierTable(); + // Noise shape: transient root (JNI local frame) -> depth-1 instance. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 6, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_JNI_LOCAL)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 7, 6, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // Real direct-retention shape: static-field root -> depth-1 instance. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 8, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 9, 8, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // Deeper chain through the transient root: passes on depth alone. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 10, 7, 2, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + // Depth-0 transient root: the candidate instance itself held by a live + // frame - suppressed like the depth-1 transient shape. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 11, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STACK_LOCAL)); + // Depth-0 durable root: the candidate instance IS the static field's + // value (the singleton-collection-itself shape) - a real direct-retention + // chain, NOT suppressible as noise. + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 12, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 7, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 9, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 10, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 11, false); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 12, false); + ASSERT_EQ(5, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag(7)) + << "depth-1 chain rooted at a transient (JNI local) root is noise"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(9)) + << "depth-1 chain rooted at a durable (static field) root is a real " + "direct-retention chain"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(10)) + << "depth-2 chain passes the gate regardless of root kind"; + EXPECT_FALSE(ReferenceChainsTestAccessor::hasResolvedChainForTag(11)) + << "depth-0 chain rooted at a transient (stack local) root is noise"; + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(12)) + << "depth-0 chain rooted at a durable (static field) root is the " + "direct-retention shape the search exists to report"; + + tracker->stop(); +} + +// Orphan slot sweep: a candidate that qualified long enough for the walk to +// record discovered instances, then stopped qualifying (its trend aged out +// of the poll's candidate list), must still get chains built for those +// instances. The slot persists by design precisely so the klass "can still +// be found there" - before the sweep, nothing iterated it once the klass +// left the poll candidates, stranding every instance recorded while it +// qualified (observed live: 8 discovered instances recorded the pass +// after the candidate's trend aged out were never built across the +// remaining 116 passes of the run). +TEST_F(PollWatchedTargetsTest, OrphanedSlotBuildsDiscoveredChainsAfterCandidateDropsOut) { + Arguments args; + ASSERT_FALSE(args.parse("referencechains=true")); + ReferenceChainTracker *tracker = ReferenceChainTracker::instance(); + ASSERT_FALSE(tracker->start(args)); + + int fake_object_storage = 0; + jobject obj = reinterpret_cast(&fake_object_storage); + seedGrowingCandidate(/*klass_id=*/3, /*rep=*/(jweak)obj); + + // First poll admits klass 3 into candidate slot 0 (nothing discovered + // yet, so nothing is built here). + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + ASSERT_EQ(0, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + // The walk discovered an instance while the candidate still + // qualified: the real direct-retention shape (static-field root -> + // depth-1 instance), which the discovered-chain gate lets through. + FrontierTable *frontier = tracker->frontierTable(); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 9, 0, 0, FrontierEntryState::EXPANDED, + JVMTI_HEAP_REFERENCE_STATIC_FIELD)); + ASSERT_TRUE(ReferenceChainsTestAccessor::insertFrontierEntry( + frontier, 8, 9, 1, FrontierEntryState::EXPANDED, /*root_kind=*/0)); + ReferenceChainsTestAccessor::recordDiscoveredInstanceForTest(3, 8, false); + ASSERT_EQ(1, ReferenceChainsTestAccessor::candidateDiscoveredCountForTest(0)); + + // The candidate stops qualifying: LivenessTracker's population table + // is wiped, so selectLeakCandidates() returns 0 on every poll from + // here on. Before the orphan sweep, this poll would leave the recorded + // instance permanently unbuildable. + LivenessTracker::instance()->klassPopulationResetForTest(); + tracker->pollWatchedTargets(&mock_jvmti, &mock_jni); + + EXPECT_TRUE(ReferenceChainsTestAccessor::hasResolvedChainForTag(8)) + << "discovered instances recorded while the candidate qualified must " + "still get chains built after it stops qualifying"; + + tracker->stop(); +}