Files
kernbench2/tests/attention/test_milestone_gqa_decode_4cases.py
T
mukesh 5672c8f3ef gqa(decode-4cases): Case 2 — Cube-Repl × PE-TP (no comm; 8× memory) (5C.B)
Second case in the GQA decode 4-cases comparative study per
GQA_full_deck.pptx slide 11. Case 2 replicates K, V across all 8
cubes × 8 PEs (the slide-11 8 KB/tok/PE memory waste) and has zero
inter-rank comm. For B=1 (single-user decode default), only PE 0
of CUBE 0 has work — the inherent PE-TP waste slide 11 calls out.

Changes:
- New kernel: src/kernbench/benches/_gqa_attention_decode_cube_repl_pe_tp.py
  Simplest of the 4 cases. Active rank loads full Q/K/V from HBM,
  computes attention via S_kv tile sweep with online-softmax merge,
  writes O. All non-(0,0) ranks early-return. No tl.send/recv.
- src/kernbench/benches/milestone_gqa_decode_4cases.py:
    - Add panel single_kv_group_decode_gqa_cube_repl_pe_tp (Case 2)
      to _PANELS and _PANEL_DISPATCH.
    - Add _run_decode_panel_cube_repl_pe_tp helper: DPPolicy K/V/Q/O
      = cube=replicate, pe=replicate (models 8× memory waste).
    - Extend _make_bench_fn to dispatch kind="decode_cube_repl_pe_tp"
      to the new runner.
- tests/attention/test_milestone_gqa_decode_4cases.py:
    4 new tests assert Case 2 contract: panel registered, smoke
    completion, zero ipcq_copy (no comm), single dma_write from cube 0.

Verification: 8/8 tests pass (4 Case 4 anchor + 4 new Case 2).

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-06-15 14:59:58 -07:00

296 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Tests for the decode 4-cases comparative-study bench.
Per ``GQA_full_deck.pptx`` slides 11-17, the 4 cases differ in how KV
cache is sharded across the 8 cubes and 8 PEs of a single KV-head
group on LLaMA-3.1-70B GQA:
Case 1 Cube-SP / PE-TP → KV split by S_kv across cubes; PEs TP on batch
Case 2 Cube-Repl / PE-TP → full KV per cube; PEs TP on batch
Case 3 Cube-Repl / PE-SP → full KV per cube; PEs SP on S_kv (intra-cube AR)
Case 4 Cube-SP / PE-SP → KV split 64-way; 2-phase AR on (m,,O) ★ optimal
This file grows as each case lands. Phase 1 of 5C.D adds Case 4 first,
which is structurally the existing ``_gqa_attention_decode_long.py`` at
``sub_w=4`` (Increment 2's lrab-adapted center-root reduce). The Phase 2
production change introduces a new bench file
``src/kernbench/benches/milestone_gqa_decode_4cases.py`` housing the 4
case panels under a single ``milestone-gqa-decode-4cases`` entry, and
extends ``_run_decode_panel`` in ``milestone_gqa_headline`` to accept
``sub_w``/``d_head``/``h_q``/``h_kv`` overrides.
Deviation from slide 13: slide prescribes AllReduce on (m,,O); the
kernel does reduce-to-root (only the lrab center cube has the answer)
per ADR-0060 §4. Treated as the kernbench Case-4 baseline.
Phase 1: tests only. T1, T2, T3, T4 fail today (new bench file does
not yet exist; ``_run_decode_panel`` does not yet accept ``sub_w``).
"""
from __future__ import annotations
import re
from pathlib import Path
from kernbench.benches.milestone_gqa_headline import _run_decode_panel
from kernbench.runtime_api.bench_runner import run_bench
from kernbench.runtime_api.types import resolve_device
from kernbench.sim_engine.engine import GraphEngine
from kernbench.topology.builder import resolve_topology
TOPOLOGY_DEFAULT = Path(__file__).resolve().parents[2] / "topology.yaml"
_CASE4_PANEL = "single_kv_group_decode_gqa_cube_sp_pe_sp"
_CASE2_PANEL = "single_kv_group_decode_gqa_cube_repl_pe_tp"
_CUBE_RE = re.compile(r"\bcube(\d+)\b")
def _engine_factory(t, d):
return GraphEngine(getattr(t, "topology_obj", t), enable_data=True)
def _count(op_log, name: str) -> int:
return sum(1 for r in op_log if r.op_name == name)
def _dma_write_cubes(op_log) -> list[int]:
cubes: list[int] = []
for r in op_log:
if r.op_name != "dma_write":
continue
m = _CUBE_RE.search(r.component_id)
if m is not None:
cubes.append(int(m.group(1)))
return cubes
def _run_case4_smoke(*, S_kv: int):
"""Drive the Case 4 decode panel via ``_run_decode_panel``.
Uses ``S_kv=8192`` (smoke) to keep test time bounded; the headline
``S_kv=128K`` runs come from ``kernbench run --bench
milestone-gqa-decode-4cases``, not pytest.
"""
topo = resolve_topology(str(TOPOLOGY_DEFAULT))
def _bench_fn(ctx):
_run_decode_panel(
ctx, panel=_CASE4_PANEL,
C=8, P=8, sub_w=4,
S_kv=S_kv,
d_head=128, h_q=8, h_kv=1,
)
return run_bench(
topology=topo, bench_fn=_bench_fn,
device=resolve_device(None),
engine_factory=_engine_factory,
)
# ── T1: Case 4 panel is registered in the new bench ──────────────────
def test_case4_panel_registered():
"""The Case 4 panel must be in the new bench's ``_PANELS`` +
``_PANEL_DISPATCH`` with the expected LLaMA-3.1-70B target dims.
Headline config:
C = 8 (head-parallel CUBE Group)
P = 8 (intra-CUBE PE-SP)
sub_w = 4 (lrab-adapted center-root reduce; root cube 6)
T_q = 1 (decode: one new token per pass)
S_kv = 131_072 (LLaMA long-context decode target)
d_head = 128, h_q = 8, h_kv = 1
"""
from kernbench.benches.milestone_gqa_decode_4cases import (
_PANEL_DISPATCH,
_PANELS,
)
assert _CASE4_PANEL in _PANELS, (
f"{_CASE4_PANEL!r} not in _PANELS; got {_PANELS}"
)
assert _CASE4_PANEL in _PANEL_DISPATCH
kind, params = _PANEL_DISPATCH[_CASE4_PANEL]
assert kind == "decode", f"kind={kind!r}, expected 'decode'"
assert params.get("C") == 8
assert params.get("P") == 8
assert params.get("sub_w") == 4
assert params.get("T_q") == 1
assert params.get("S_kv") == 131_072
assert params.get("d_head") == 128
assert params.get("h_q") == 8
assert params.get("h_kv") == 1
# ── T2: Case 4 runner drives the kernel to completion ───────────────
def test_case4_runner_smoke():
"""``_run_decode_panel`` must accept the new ``sub_w``,
``d_head``, ``h_q``, ``h_kv`` kwargs and launch the kernel at
``(C, P, sub_w) = (8, 8, 4)`` (the Case 4 / lrab path).
Smoke uses ``S_kv=8192`` so the simulation completes quickly. The
headline 128K dims run via ``kernbench run --bench
milestone-gqa-decode-4cases``.
"""
result = _run_case4_smoke(S_kv=8192)
assert result.completion.ok, (
f"Case 4 decode smoke at C=8 P=8 sub_w=4 must complete; "
f"got {result.completion}"
)
# ── T3: reduce-to-root lands at the lrab center cube (cube 6) ───────
def test_case4_root_at_center_cube_6():
"""For ``sub_w=4, sub_h=2``: root_col=2, root_row=1, root_cube=6.
The decode kernel writes the final O exclusively from PE 0 of cube
6 (ADR-0060 §4 reduce-to-root variant of the Case-4 AR pattern).
"""
result = _run_case4_smoke(S_kv=8192)
assert result.completion.ok
cubes = _dma_write_cubes(result.engine.op_log)
assert cubes, "expected at least one dma_write for the final O store"
distinct = set(cubes)
assert distinct == {6}, (
f"Case 4 root must be the lrab center cube 6; "
f"got cubes={sorted(distinct)}"
)
# ── T4: 2-phase AR ipcq pattern matches the predicted Case-4 traffic ─
def _run_case2_smoke(*, S_kv: int):
"""Drive the Case 2 decode panel via the case-specific runner.
Case 2 = Cube-Repl × PE-TP. K, V are replicated everywhere (the
slide-11 memory waste); for B=1 only one rank does the work; no
inter-rank comm.
"""
from kernbench.benches.milestone_gqa_decode_4cases import (
_run_decode_panel_cube_repl_pe_tp,
)
topo = resolve_topology(str(TOPOLOGY_DEFAULT))
def _bench_fn(ctx):
_run_decode_panel_cube_repl_pe_tp(
ctx, panel=_CASE2_PANEL,
C=8, P=8,
T_q=1, S_kv=S_kv,
d_head=128, h_q=8, h_kv=1,
)
return run_bench(
topology=topo, bench_fn=_bench_fn,
device=resolve_device(None),
engine_factory=_engine_factory,
)
# ── Case 2 — T1: panel registered ───────────────────────────────────
def test_case2_panel_registered():
"""The Case 2 panel must be in the bench's ``_PANELS`` +
``_PANEL_DISPATCH`` with the expected single-KV-group dims.
Case 2: Cube-Repl × PE-TP. K, V replicated everywhere
(8 KB/tok/PE — slide-11 memory waste); no inter-rank comm.
For B=1 only one rank works (PEs 1-7 idle — slide-11 calls
out this PE-TP waste).
"""
from kernbench.benches.milestone_gqa_decode_4cases import (
_PANEL_DISPATCH,
_PANELS,
)
assert _CASE2_PANEL in _PANELS, (
f"{_CASE2_PANEL!r} not in _PANELS; got {_PANELS}"
)
assert _CASE2_PANEL in _PANEL_DISPATCH
kind, params = _PANEL_DISPATCH[_CASE2_PANEL]
assert kind == "decode_cube_repl_pe_tp"
assert params.get("C") == 8
assert params.get("P") == 8
assert params.get("T_q") == 1
assert params.get("S_kv") == 131_072
assert params.get("d_head") == 128
assert params.get("h_q") == 8
assert params.get("h_kv") == 1
# ── Case 2 — T2: smoke runner completes ─────────────────────────────
def test_case2_runner_smoke():
"""Case 2 runner drives the new kernel to completion at smoke S_kv."""
result = _run_case2_smoke(S_kv=8192)
assert result.completion.ok, (
f"Case 2 decode smoke at C=8 P=8 must complete; "
f"got {result.completion}"
)
# ── Case 2 — T3: zero inter-rank comm by design ─────────────────────
def test_case2_zero_ipcq_copy_no_comm():
"""Case 2's defining property: full KV per rank ⇒ NO inter-rank
communication. Slide 11 lists comm cost as 'none'.
"""
result = _run_case2_smoke(S_kv=8192)
assert result.completion.ok
n_copy = _count(result.engine.op_log, "ipcq_copy")
assert n_copy == 0, (
f"Case 2 must have zero inter-rank comm; got ipcq_copy={n_copy}"
)
# ── Case 2 — T4: single dma_write from cube 0 (B=1 single-rank work) ─
def test_case2_single_dma_write_at_cube_0():
"""For B=1, only PE 0 of CUBE 0 does the work (the inherent PE-TP
waste at B=1). Exactly 1 dma_write, from cube 0.
"""
result = _run_case2_smoke(S_kv=8192)
assert result.completion.ok
cubes = _dma_write_cubes(result.engine.op_log)
assert cubes, "expected at least one dma_write for the final O store"
distinct = set(cubes)
assert distinct == {0}, (
f"Case 2 B=1 single writer must be cube 0; "
f"got cubes={sorted(distinct)}"
)
# ── Case 4 — T4 (existing, kept) ────────────────────────────────────
def test_case4_two_level_ar_ipcq_pattern():
"""Total ipcq_copy for the Case 4 reduce at (C, P, sub_w) =
(8, 8, 4):
Intra-CUBE (per CUBE = 8 PEs in a 2×4 grid):
row chain along intra_W: cols 1,2,3 each row × 2 rows ×
3 tensors = 18
col bridge along intra_N: pe4 only × 3 tensors = 3
per-CUBE intra total = 21
× 8 CUBEs = 168
Inter-CUBE lrab (sub_w=4, sub_h=2):
Phase 1 row reduce — 3 sends/row × 3 tensors × 2 rows = 18
Phase 2 col reduce — cube 2 → S × 3 tensors = 3
inter-CUBE total = 21
Grand total: 168 + 21 = 189
"""
result = _run_case4_smoke(S_kv=8192)
assert result.completion.ok
n_copy = _count(result.engine.op_log, "ipcq_copy")
assert n_copy == 189, (
f"Case 4 expected 189 ipcq_copy "
f"(168 intra-CUBE + 21 inter-CUBE lrab); got {n_copy}"
)