policy: add DPPolicy.cube_start for disjoint cube sub-meshes within a SIP
Adds an optional ``cube_start: int = 0`` field to ``DPPolicy``. In ``resolve_dp_policy`` the generated ``ShardSpec.cube`` is now ``policy.cube_start + cube_id`` instead of just ``cube_id``. With the default value (0) every existing call site resolves to identical ``ShardSpec.cube`` values (cubes 0..num_cubes-1). Why this is needed: the GQA Llama-70B 8-KV-group headline target lays two 2x4 KV-groups per SIP — the first on cubes 0..7 (rows 0..1), the second on cubes 8..15 (rows 2..3). Without ``cube_start``, ``DPPolicy`` can only address the first 8 row-major cubes, so a second launch on the same SIP overlaps the first. ``cube_start=8`` selects the second 2x4 sub-mesh directly. Design choice: scalar ``cube_start`` (vs a 2D ``cube_mesh_origin`` or arbitrary ``cube_ids`` list) was picked because (a) the 2D mesh-mlo kernel already assumes row-major contiguous cubes, (b) it pairs naturally with ``num_cubes`` (range = ``[start, start+count)``), and (c) zero migration churn — every existing call site stays unchanged. Scope: 5 production LOC (one field + one expression in resolve). No kernel changes here; kernels that consume ``program_id(axis=1)`` for ring arithmetic need their own follow-up (see ADR-0022 contract). Tests: 7 new unit tests in ``test_dppolicy_cube_start.py`` covering default behavior preservation (cube_start=0), shifted ranges (cube_start=8 → cubes 8..15), full-SIP CCL pattern, replicate cube policy, shard-count preservation, and uniqueness. ADR-0026 regression suite (12 tests) stays green. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -33,12 +33,20 @@ class DPPolicy:
|
|||||||
Optional overrides (``None`` = use topology dimensions):
|
Optional overrides (``None`` = use topology dimensions):
|
||||||
- num_pes: override PEs per cube
|
- num_pes: override PEs per cube
|
||||||
- num_cubes: override cubes per SIP
|
- num_cubes: override cubes per SIP
|
||||||
|
|
||||||
|
Cube range:
|
||||||
|
- cube_start: index of the first cube in the SIP to place shards
|
||||||
|
on. Defaults to 0. Enables disjoint sub-meshes within one SIP
|
||||||
|
(e.g. cubes 8..15 alongside cubes 0..7) — required by the
|
||||||
|
GQA Llama-70B 8-KV-group headline target (two 2×4 KV-groups
|
||||||
|
per SIP).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
cube: Literal["replicate", "column_wise", "row_wise"] = "replicate"
|
cube: Literal["replicate", "column_wise", "row_wise"] = "replicate"
|
||||||
pe: Literal["replicate", "column_wise", "row_wise"] = "replicate"
|
pe: Literal["replicate", "column_wise", "row_wise"] = "replicate"
|
||||||
num_pes: int | None = None
|
num_pes: int | None = None
|
||||||
num_cubes: int | None = None
|
num_cubes: int | None = None
|
||||||
|
cube_start: int = 0
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
@dataclass(frozen=True)
|
||||||
@@ -125,7 +133,7 @@ def resolve_dp_policy(
|
|||||||
for ls in local_shards:
|
for ls in local_shards:
|
||||||
all_shards.append(ShardSpec(
|
all_shards.append(ShardSpec(
|
||||||
sip=target_sip,
|
sip=target_sip,
|
||||||
cube=cube_id,
|
cube=policy.cube_start + cube_id,
|
||||||
pe=ls.local_pe,
|
pe=ls.local_pe,
|
||||||
offset_bytes=cube_offset + ls.offset_bytes,
|
offset_bytes=cube_offset + ls.offset_bytes,
|
||||||
nbytes=ls.nbytes,
|
nbytes=ls.nbytes,
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
"""Phase 1 spec test for ``DPPolicy.cube_start``.
|
||||||
|
|
||||||
|
``cube_start`` is an optional override that shifts the cube indices in
|
||||||
|
``ShardSpec`` generated by ``resolve_dp_policy``. It enables addressing a
|
||||||
|
disjoint cube sub-mesh within one SIP (e.g. cubes 8..15 in addition to
|
||||||
|
0..7), which the GQA Llama-70B 8-KV-group headline target requires
|
||||||
|
(two 2×4 KV-groups per SIP × 4 SIPs = 64 cubes).
|
||||||
|
|
||||||
|
Default ``cube_start=0`` preserves every existing call-site bit-for-bit:
|
||||||
|
``DPPolicy(num_cubes=8)`` still resolves to ``ShardSpec.cube ∈ [0, 8)``.
|
||||||
|
The CCL milestone bench's full-SIP ``DPPolicy(num_cubes=16)`` likewise
|
||||||
|
continues to produce cubes 0..15.
|
||||||
|
|
||||||
|
Phase 1: production code is unchanged → these tests FAIL until the
|
||||||
|
Phase 2 diff lands. Phase 2 makes all of them pass.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from kernbench.policy.placement.dp import DPPolicy, resolve_dp_policy
|
||||||
|
|
||||||
|
|
||||||
|
# ── Default ``cube_start=0`` preserves existing behavior ─────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_dppolicy_default_cube_start_is_zero():
|
||||||
|
"""``DPPolicy(num_cubes=8)`` sets ``cube_start=0`` by default."""
|
||||||
|
dp = DPPolicy(cube="row_wise", num_cubes=8)
|
||||||
|
assert dp.cube_start == 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_default_yields_cubes_zero_through_seven():
|
||||||
|
"""Backward-compat: default ``cube_start=0`` → cubes 0..num_cubes-1."""
|
||||||
|
dp = DPPolicy(cube="row_wise", pe="replicate", num_cubes=8, num_pes=1)
|
||||||
|
shards = resolve_dp_policy(
|
||||||
|
dp, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=8, target_sip=0,
|
||||||
|
)
|
||||||
|
cube_ids = sorted({s.cube for s in shards})
|
||||||
|
assert cube_ids == list(range(8))
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_full_sip_default_yields_cubes_zero_through_fifteen():
|
||||||
|
"""CCL pattern: ``num_cubes=16, cube_start=0`` → cubes 0..15 unchanged."""
|
||||||
|
dp = DPPolicy(cube="row_wise", pe="replicate", num_cubes=16, num_pes=1)
|
||||||
|
shards = resolve_dp_policy(
|
||||||
|
dp, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=16, target_sip=0,
|
||||||
|
)
|
||||||
|
cube_ids = sorted({s.cube for s in shards})
|
||||||
|
assert cube_ids == list(range(16))
|
||||||
|
|
||||||
|
|
||||||
|
# ── ``cube_start=N`` shifts cube indices ─────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_cube_start_eight_yields_cubes_eight_through_fifteen():
|
||||||
|
"""``cube_start=8, num_cubes=8`` → ``ShardSpec.cube ∈ [8, 16)``.
|
||||||
|
|
||||||
|
This is the headline-enabling case: the second 2×4 KV-group per SIP
|
||||||
|
in the 8-KV-group Llama-70B target lands on cubes 8..15.
|
||||||
|
"""
|
||||||
|
dp = DPPolicy(
|
||||||
|
cube="row_wise", pe="replicate",
|
||||||
|
num_cubes=8, num_pes=1, cube_start=8,
|
||||||
|
)
|
||||||
|
shards = resolve_dp_policy(
|
||||||
|
dp, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=8, target_sip=0,
|
||||||
|
)
|
||||||
|
cube_ids = sorted({s.cube for s in shards})
|
||||||
|
assert cube_ids == list(range(8, 16))
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_cube_start_preserves_shard_count():
|
||||||
|
"""Shifting ``cube_start`` does not change the total shard count."""
|
||||||
|
dp_default = DPPolicy(
|
||||||
|
cube="row_wise", pe="replicate", num_cubes=8, num_pes=1,
|
||||||
|
)
|
||||||
|
dp_shifted = DPPolicy(
|
||||||
|
cube="row_wise", pe="replicate",
|
||||||
|
num_cubes=8, num_pes=1, cube_start=8,
|
||||||
|
)
|
||||||
|
shards_def = resolve_dp_policy(
|
||||||
|
dp_default, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=8, target_sip=0,
|
||||||
|
)
|
||||||
|
shards_shf = resolve_dp_policy(
|
||||||
|
dp_shifted, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=8, target_sip=0,
|
||||||
|
)
|
||||||
|
assert len(shards_def) == len(shards_shf)
|
||||||
|
|
||||||
|
|
||||||
|
# ── ``cube_start`` works with ``cube="replicate"`` too ──────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_cube_start_with_replicate_policy():
|
||||||
|
"""``cube="replicate", cube_start=8`` → every cube in [8, 16) gets a copy."""
|
||||||
|
dp = DPPolicy(
|
||||||
|
cube="replicate", pe="replicate",
|
||||||
|
num_cubes=8, num_pes=1, cube_start=8,
|
||||||
|
)
|
||||||
|
shards = resolve_dp_policy(
|
||||||
|
dp, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=8, target_sip=0,
|
||||||
|
)
|
||||||
|
cube_ids = sorted({s.cube for s in shards})
|
||||||
|
assert cube_ids == list(range(8, 16))
|
||||||
|
|
||||||
|
|
||||||
|
# ── No duplicates regardless of ``cube_start`` ───────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def test_resolve_cube_start_yields_distinct_cube_ids():
|
||||||
|
"""For ``pe="replicate", num_pes=1``, every shard has a distinct cube."""
|
||||||
|
dp = DPPolicy(
|
||||||
|
cube="row_wise", pe="replicate",
|
||||||
|
num_cubes=8, num_pes=1, cube_start=8,
|
||||||
|
)
|
||||||
|
shards = resolve_dp_policy(
|
||||||
|
dp, shape=(32, 64), itemsize=2,
|
||||||
|
num_pe=1, num_cubes=8, target_sip=0,
|
||||||
|
)
|
||||||
|
cube_ids = [s.cube for s in shards]
|
||||||
|
assert len(cube_ids) == len(set(cube_ids))
|
||||||
|
assert all(8 <= c < 16 for c in cube_ids)
|
||||||
Reference in New Issue
Block a user