From 51b8223d9aaa7a71398b771f537e592d03ad2ff2 Mon Sep 17 00:00:00 2001 From: Hao Wu Date: Fri, 2 Oct 2026 20:49:45 -0400 Subject: [PATCH] [FIX] Report the profiler's buffer-load issue only for AMD kernels The buffer-load check is meant for AMD GPUs, but has_buffer_load started as False and was only updated from an amdgcn ASM stage. On NVIDIA no amdgcn stage exists, so every kernel whose offsets fit in 32 bits was flagged as a potential buffer-load issue and the summary printed the AMD warning. has_buffer_load now starts as None and is set only from a captured amdgcn stage, so the existing `is False` test flags only AMD kernels without buffer_load. When no amdgcn kernel was captured, the summary says the check was skipped. Unit tests cover the NVIDIA-like, AMD without buffer_load and AMD with buffer_load cases. --- tests/unit/test_profiler.py | 47 +++++++++++++++++++++++++++ tilelens/clients/profiler/profiler.py | 8 +++-- 2 files changed, 53 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_profiler.py b/tests/unit/test_profiler.py index 23f303d97..384e73082 100644 --- a/tests/unit/test_profiler.py +++ b/tests/unit/test_profiler.py @@ -1,4 +1,5 @@ import numpy as np +from types import SimpleNamespace from unittest.mock import patch import pytest @@ -272,6 +273,52 @@ def test_check_32bit_range_no_buffer_load(_isolate_profiler_cfg): assert profiler.potential_buffer_load_issue_found is True +# ======== Buffer Load Check: AMD-only Tests ========= + + +def _warmup_then_check_32bit(asm: dict[str, str]) -> Profiler: + """Feed a fake warmup result with ``asm`` stages, then 32-bit-range offsets.""" + cfg.profiler_enable_block_sampling = False + cfg.profiler_enable_load_store_skipping = False + cfg.profiler_disable_buffer_load_check = False + + profiler = Profiler() + profiler.post_warmup_callback(None, SimpleNamespace(asm=asm)) + + byte_offset = np.array([0, 1000, 2000]) + profiler._check_32bit_range(byte_offset, 4, byte_offset // 4) + return profiler + + +def test_buffer_load_check_skipped_without_amdgcn(_isolate_profiler_cfg, capsys): + """An NVIDIA-like warmup result (no amdgcn stage) never flags a buffer-load issue.""" + profiler = _warmup_then_check_32bit({"ttir": "tt.load", "ptx": "ld.global.f32"}) + + assert profiler.has_buffer_load is None + assert profiler.potential_buffer_load_issue_found is False + + profiler.finalize() + out = capsys.readouterr().out + assert "Buffer Load check skipped" in out + assert "Potential Buffer Load Issue" not in out + + +def test_buffer_load_check_flags_amdgcn_without_buffer_load(_isolate_profiler_cfg): + """An amdgcn stage without buffer_load and 32-bit offsets is flagged.""" + profiler = _warmup_then_check_32bit({"amdgcn": "global_load_dword v0, v[0:1]"}) + + assert profiler.has_buffer_load is False + assert profiler.potential_buffer_load_issue_found is True + + +def test_buffer_load_check_passes_amdgcn_with_buffer_load(_isolate_profiler_cfg): + """An amdgcn stage that uses buffer_load is not flagged.""" + profiler = _warmup_then_check_32bit({"amdgcn": "buffer_load_dword v0, v1, s[0:3]"}) + + assert profiler.has_buffer_load is True + assert profiler.potential_buffer_load_issue_found is False + + # ======== Pre-run Callback Tests ========= diff --git a/tilelens/clients/profiler/profiler.py b/tilelens/clients/profiler/profiler.py index e95723549..7b89cf92b 100644 --- a/tilelens/clients/profiler/profiler.py +++ b/tilelens/clients/profiler/profiler.py @@ -79,7 +79,8 @@ def __init__( self.mask_op_stats: list[MaskOpStats] = [] # Case 4: Buffer Load Check - self.has_buffer_load = False + # None until an AMD (amdgcn) kernel ASM is captured; the check only applies there. + self.has_buffer_load: bool | None = None self.disable_buffer_load_check = cfg.profiler_disable_buffer_load_check self.potential_buffer_load_issue_found = False @@ -195,6 +196,7 @@ def _check_32bit_range( if num_outside == 0: # All offsets are within 32-bit range # If we're on AMD GPU and buffer_load is NOT found, this is an error + # (has_buffer_load stays None when no amdgcn stage was captured, e.g. NVIDIA) if self.has_buffer_load is False: # Buffer Load optimization should be used when offsets are within 32-bit range. self.potential_buffer_load_issue_found = True @@ -516,7 +518,9 @@ def finalize(self) -> list: + "-" * 11 ) print("=" * 60) - if self.potential_buffer_load_issue_found: + if self.has_buffer_load is None: + print("Buffer Load check skipped: no AMD (amdgcn) kernel was captured.") + elif self.potential_buffer_load_issue_found: print("\n>>>>>> Warning: Potential Buffer Load Issue Detected! <<<<<<") print( "\nSome memory access offsets are within 32-bit range, "