diff --git a/tests/unit/test_profiler.py b/tests/unit/test_profiler.py index 23f303d97..384e73082 100644 --- a/tests/unit/test_profiler.py +++ b/tests/unit/test_profiler.py @@ -1,4 +1,5 @@ import numpy as np +from types import SimpleNamespace from unittest.mock import patch import pytest @@ -272,6 +273,52 @@ def test_check_32bit_range_no_buffer_load(_isolate_profiler_cfg): assert profiler.potential_buffer_load_issue_found is True +# ======== Buffer Load Check: AMD-only Tests ========= + + +def _warmup_then_check_32bit(asm: dict[str, str]) -> Profiler: + """Feed a fake warmup result with ``asm`` stages, then 32-bit-range offsets.""" + cfg.profiler_enable_block_sampling = False + cfg.profiler_enable_load_store_skipping = False + cfg.profiler_disable_buffer_load_check = False + + profiler = Profiler() + profiler.post_warmup_callback(None, SimpleNamespace(asm=asm)) + + byte_offset = np.array([0, 1000, 2000]) + profiler._check_32bit_range(byte_offset, 4, byte_offset // 4) + return profiler + + +def test_buffer_load_check_skipped_without_amdgcn(_isolate_profiler_cfg, capsys): + """An NVIDIA-like warmup result (no amdgcn stage) never flags a buffer-load issue.""" + profiler = _warmup_then_check_32bit({"ttir": "tt.load", "ptx": "ld.global.f32"}) + + assert profiler.has_buffer_load is None + assert profiler.potential_buffer_load_issue_found is False + + profiler.finalize() + out = capsys.readouterr().out + assert "Buffer Load check skipped" in out + assert "Potential Buffer Load Issue" not in out + + +def test_buffer_load_check_flags_amdgcn_without_buffer_load(_isolate_profiler_cfg): + """An amdgcn stage without buffer_load and 32-bit offsets is flagged.""" + profiler = _warmup_then_check_32bit({"amdgcn": "global_load_dword v0, v[0:1]"}) + + assert profiler.has_buffer_load is False + assert profiler.potential_buffer_load_issue_found is True + + +def test_buffer_load_check_passes_amdgcn_with_buffer_load(_isolate_profiler_cfg): + """An amdgcn stage that uses buffer_load is not flagged.""" + profiler = _warmup_then_check_32bit({"amdgcn": "buffer_load_dword v0, v1, s[0:3]"}) + + assert profiler.has_buffer_load is True + assert profiler.potential_buffer_load_issue_found is False + + # ======== Pre-run Callback Tests ========= diff --git a/tilelens/clients/profiler/profiler.py b/tilelens/clients/profiler/profiler.py index e95723549..7b89cf92b 100644 --- a/tilelens/clients/profiler/profiler.py +++ b/tilelens/clients/profiler/profiler.py @@ -79,7 +79,8 @@ def __init__( self.mask_op_stats: list[MaskOpStats] = [] # Case 4: Buffer Load Check - self.has_buffer_load = False + # None until an AMD (amdgcn) kernel ASM is captured; the check only applies there. + self.has_buffer_load: bool | None = None self.disable_buffer_load_check = cfg.profiler_disable_buffer_load_check self.potential_buffer_load_issue_found = False @@ -195,6 +196,7 @@ def _check_32bit_range( if num_outside == 0: # All offsets are within 32-bit range # If we're on AMD GPU and buffer_load is NOT found, this is an error + # (has_buffer_load stays None when no amdgcn stage was captured, e.g. NVIDIA) if self.has_buffer_load is False: # Buffer Load optimization should be used when offsets are within 32-bit range. self.potential_buffer_load_issue_found = True @@ -516,7 +518,9 @@ def finalize(self) -> list: + "-" * 11 ) print("=" * 60) - if self.potential_buffer_load_issue_found: + if self.has_buffer_load is None: + print("Buffer Load check skipped: no AMD (amdgcn) kernel was captured.") + elif self.potential_buffer_load_issue_found: print("\n>>>>>> Warning: Potential Buffer Load Issue Detected! <<<<<<") print( "\nSome memory access offsets are within 32-bit range, "