Regenerate latency plots/diagrams for post-Phase-2c model

Allreduce + pe2pe + ipcq + pe_view auto-regenerated by test sweeps
running against the new chunk-streaming wire timing (per-flit
wormhole) — absolute numbers shift upward to reflect bottleneck-link
transit charged once per flit (instead of the previous cut-through
subtraction at HBM CTRL).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-05-14 23:24:01 -07:00
parent a0cccc71e8
commit a44f832be5
17 changed files with 231 additions and 163 deletions
@@ -1,13 +1,13 @@
buffer_kind,sip_topology,n_sips,n_elem,bytes_per_pe,latency_ns
hbm,torus_2d,6,128,256,1858.0399999999827
hbm,torus_2d,6,1024,2048,2389.0399999999827
hbm,torus_2d,6,8192,16384,6673.039999999986
hbm,torus_2d,6,32768,65536,21361.03999999992
sram,torus_2d,6,128,256,1774.0399999999827
sram,torus_2d,6,1024,2048,2389.0399999999827
sram,torus_2d,6,8192,16384,7345.039999999986
sram,torus_2d,6,32768,65536,24337.039999999935
tcm,torus_2d,6,128,256,1678.0399999999827
tcm,torus_2d,6,1024,2048,1957.0399999999827
tcm,torus_2d,6,8192,16384,4225.039999999986
tcm,torus_2d,6,32768,65536,12001.03999999992
buffer_kind,sip_topology,n_sips,n_elem,bytes_per_pe,latency_ns
hbm,torus_2d,6,128,256,2144.0399999999754
hbm,torus_2d,6,1024,2048,2908.74499999995
hbm,torus_2d,6,8192,16384,8851.185000000081
hbm,torus_2d,6,32768,65536,29225.265000008752
sram,torus_2d,6,128,256,2060.0399999999754
sram,torus_2d,6,1024,2048,2908.74499999995
sram,torus_2d,6,8192,16384,9523.185000000081
sram,torus_2d,6,32768,65536,32201.265000008752
tcm,torus_2d,6,128,256,1964.0399999999754
tcm,torus_2d,6,1024,2048,2476.74499999995
tcm,torus_2d,6,8192,16384,6403.185000000081
tcm,torus_2d,6,32768,65536,19865.265000008738
1 buffer_kind sip_topology n_sips n_elem bytes_per_pe latency_ns
2 hbm torus_2d 6 128 256 1858.0399999999827 2144.0399999999754
3 hbm torus_2d 6 1024 2048 2389.0399999999827 2908.74499999995
4 hbm torus_2d 6 8192 16384 6673.039999999986 8851.185000000081
5 hbm torus_2d 6 32768 65536 21361.03999999992 29225.265000008752
6 sram torus_2d 6 128 256 1774.0399999999827 2060.0399999999754
7 sram torus_2d 6 1024 2048 2389.0399999999827 2908.74499999995
8 sram torus_2d 6 8192 16384 7345.039999999986 9523.185000000081
9 sram torus_2d 6 32768 65536 24337.039999999935 32201.265000008752
10 tcm torus_2d 6 128 256 1678.0399999999827 1964.0399999999754
11 tcm torus_2d 6 1024 2048 1957.0399999999827 2476.74499999995
12 tcm torus_2d 6 8192 16384 4225.039999999986 6403.185000000081
13 tcm torus_2d 6 32768 65536 12001.03999999992 19865.265000008738
Binary file not shown.

Before

Width:  |  Height:  |  Size: 74 KiB

After

Width:  |  Height:  |  Size: 76 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 40 KiB

After

Width:  |  Height:  |  Size: 39 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 82 KiB

After

Width:  |  Height:  |  Size: 79 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 38 KiB

After

Width:  |  Height:  |  Size: 37 KiB

@@ -1,37 +1,37 @@
algorithm,sip_topology,n_sips,n_elem,bytes_per_pe,bytes_per_sip,latency_ns
intercube_allreduce,mesh_2d_no_wrap,6,8,16,256,2626.302499999998
intercube_allreduce,mesh_2d_no_wrap,6,32,64,1024,2634.7399999999952
intercube_allreduce,mesh_2d_no_wrap,6,64,128,2048,2645.9899999999925
intercube_allreduce,mesh_2d_no_wrap,6,128,256,4096,2668.489999999987
intercube_allreduce,mesh_2d_no_wrap,6,512,1024,16384,2812.489999999987
intercube_allreduce,mesh_2d_no_wrap,6,1024,2048,32768,3010.489999999987
intercube_allreduce,mesh_2d_no_wrap,6,2048,4096,65536,3406.489999999987
intercube_allreduce,mesh_2d_no_wrap,6,4096,8192,131072,4198.489999999965
intercube_allreduce,mesh_2d_no_wrap,6,8192,16384,262144,5782.489999999969
intercube_allreduce,mesh_2d_no_wrap,6,16384,32768,524288,8950.489999999925
intercube_allreduce,mesh_2d_no_wrap,6,32768,65536,1048576,15286.48999999986
intercube_allreduce,mesh_2d_no_wrap,6,49152,98304,1572864,21622.489999999932
intercube_allreduce,ring_1d,6,8,16,256,2302.9849999999933
intercube_allreduce,ring_1d,6,32,64,1024,2310.8599999999906
intercube_allreduce,ring_1d,6,64,128,2048,2321.359999999988
intercube_allreduce,ring_1d,6,128,256,4096,2342.3599999999824
intercube_allreduce,ring_1d,6,512,1024,16384,2479.3599999999824
intercube_allreduce,ring_1d,6,1024,2048,32768,2669.3599999999824
intercube_allreduce,ring_1d,6,2048,4096,65536,3049.3599999999824
intercube_allreduce,ring_1d,6,4096,8192,131072,3809.3599999999715
intercube_allreduce,ring_1d,6,8192,16384,262144,5329.359999999979
intercube_allreduce,ring_1d,6,16384,32768,524288,8369.35999999992
intercube_allreduce,ring_1d,6,32768,65536,1048576,14449.359999999899
intercube_allreduce,ring_1d,6,49152,98304,1572864,20529.35999999997
intercube_allreduce,torus_2d,6,8,16,256,1644.2899999999936
intercube_allreduce,torus_2d,6,32,64,1024,1651.0399999999909
intercube_allreduce,torus_2d,6,64,128,2048,1660.0399999999881
intercube_allreduce,torus_2d,6,128,256,4096,1678.0399999999827
intercube_allreduce,torus_2d,6,512,1024,16384,1795.0399999999827
intercube_allreduce,torus_2d,6,1024,2048,32768,1957.0399999999827
intercube_allreduce,torus_2d,6,2048,4096,65536,2281.0399999999827
intercube_allreduce,torus_2d,6,4096,8192,131072,2929.039999999979
intercube_allreduce,torus_2d,6,8192,16384,262144,4225.039999999986
intercube_allreduce,torus_2d,6,16384,32768,524288,6817.039999999943
intercube_allreduce,torus_2d,6,32768,65536,1048576,12001.03999999992
intercube_allreduce,torus_2d,6,49152,98304,1572864,17185.039999999994
algorithm,sip_topology,n_sips,n_elem,bytes_per_pe,bytes_per_sip,latency_ns
intercube_allreduce,mesh_2d_no_wrap,6,8,16,256,2666.5524999999725
intercube_allreduce,mesh_2d_no_wrap,6,32,64,1024,2747.7399999999725
intercube_allreduce,mesh_2d_no_wrap,6,64,128,2048,2855.98999999998
intercube_allreduce,mesh_2d_no_wrap,6,128,256,4096,3072.4899999999725
intercube_allreduce,mesh_2d_no_wrap,6,512,1024,16384,3336.579999999951
intercube_allreduce,mesh_2d_no_wrap,6,1024,2048,32768,3707.49999999992
intercube_allreduce,mesh_2d_no_wrap,6,2048,4096,65536,4449.339999999875
intercube_allreduce,mesh_2d_no_wrap,6,4096,8192,131072,5933.020000000055
intercube_allreduce,mesh_2d_no_wrap,6,8192,16384,262144,8900.380000000157
intercube_allreduce,mesh_2d_no_wrap,6,16384,32768,524288,14835.099999997583
intercube_allreduce,mesh_2d_no_wrap,6,32768,65536,1048576,26704.540000017492
intercube_allreduce,mesh_2d_no_wrap,6,49152,98304,1572864,38573.980000026335
intercube_allreduce,ring_1d,6,8,16,256,2365.2558333333036
intercube_allreduce,ring_1d,6,32,64,1024,2436.9433333333036
intercube_allreduce,ring_1d,6,64,128,2048,2532.526666666643
intercube_allreduce,ring_1d,6,128,256,4096,2723.6933333333036
intercube_allreduce,ring_1d,6,512,1024,16384,3042.0349999999544
intercube_allreduce,ring_1d,6,1024,2048,32768,3390.201666666597
intercube_allreduce,ring_1d,6,2048,4096,65536,4079.7349999998714
intercube_allreduce,ring_1d,6,4096,8192,131072,5458.801666666721
intercube_allreduce,ring_1d,6,8192,16384,262144,8216.93500000014
intercube_allreduce,ring_1d,6,16384,32768,524288,13733.201666664638
intercube_allreduce,ring_1d,6,32768,65536,1048576,24765.735000014545
intercube_allreduce,ring_1d,6,49152,98304,1572864,35798.268333355256
intercube_allreduce,torus_2d,6,8,16,256,1700.6024999999754
intercube_allreduce,torus_2d,6,32,64,1024,1753.2899999999754
intercube_allreduce,torus_2d,6,64,128,2048,1823.539999999979
intercube_allreduce,torus_2d,6,128,256,4096,1964.0399999999754
intercube_allreduce,torus_2d,6,512,1024,16384,2196.2849999999653
intercube_allreduce,torus_2d,6,1024,2048,32768,2476.74499999995
intercube_allreduce,torus_2d,6,2048,4096,65536,3037.664999999919
intercube_allreduce,torus_2d,6,4096,8192,131072,4159.50500000003
intercube_allreduce,torus_2d,6,8192,16384,262144,6403.185000000081
intercube_allreduce,torus_2d,6,16384,32768,524288,10890.544999998769
intercube_allreduce,torus_2d,6,32768,65536,1048576,19865.265000008738
intercube_allreduce,torus_2d,6,49152,98304,1572864,28839.985000013185
1 algorithm sip_topology n_sips n_elem bytes_per_pe bytes_per_sip latency_ns
2 intercube_allreduce mesh_2d_no_wrap 6 8 16 256 2626.302499999998 2666.5524999999725
3 intercube_allreduce mesh_2d_no_wrap 6 32 64 1024 2634.7399999999952 2747.7399999999725
4 intercube_allreduce mesh_2d_no_wrap 6 64 128 2048 2645.9899999999925 2855.98999999998
5 intercube_allreduce mesh_2d_no_wrap 6 128 256 4096 2668.489999999987 3072.4899999999725
6 intercube_allreduce mesh_2d_no_wrap 6 512 1024 16384 2812.489999999987 3336.579999999951
7 intercube_allreduce mesh_2d_no_wrap 6 1024 2048 32768 3010.489999999987 3707.49999999992
8 intercube_allreduce mesh_2d_no_wrap 6 2048 4096 65536 3406.489999999987 4449.339999999875
9 intercube_allreduce mesh_2d_no_wrap 6 4096 8192 131072 4198.489999999965 5933.020000000055
10 intercube_allreduce mesh_2d_no_wrap 6 8192 16384 262144 5782.489999999969 8900.380000000157
11 intercube_allreduce mesh_2d_no_wrap 6 16384 32768 524288 8950.489999999925 14835.099999997583
12 intercube_allreduce mesh_2d_no_wrap 6 32768 65536 1048576 15286.48999999986 26704.540000017492
13 intercube_allreduce mesh_2d_no_wrap 6 49152 98304 1572864 21622.489999999932 38573.980000026335
14 intercube_allreduce ring_1d 6 8 16 256 2302.9849999999933 2365.2558333333036
15 intercube_allreduce ring_1d 6 32 64 1024 2310.8599999999906 2436.9433333333036
16 intercube_allreduce ring_1d 6 64 128 2048 2321.359999999988 2532.526666666643
17 intercube_allreduce ring_1d 6 128 256 4096 2342.3599999999824 2723.6933333333036
18 intercube_allreduce ring_1d 6 512 1024 16384 2479.3599999999824 3042.0349999999544
19 intercube_allreduce ring_1d 6 1024 2048 32768 2669.3599999999824 3390.201666666597
20 intercube_allreduce ring_1d 6 2048 4096 65536 3049.3599999999824 4079.7349999998714
21 intercube_allreduce ring_1d 6 4096 8192 131072 3809.3599999999715 5458.801666666721
22 intercube_allreduce ring_1d 6 8192 16384 262144 5329.359999999979 8216.93500000014
23 intercube_allreduce ring_1d 6 16384 32768 524288 8369.35999999992 13733.201666664638
24 intercube_allreduce ring_1d 6 32768 65536 1048576 14449.359999999899 24765.735000014545
25 intercube_allreduce ring_1d 6 49152 98304 1572864 20529.35999999997 35798.268333355256
26 intercube_allreduce torus_2d 6 8 16 256 1644.2899999999936 1700.6024999999754
27 intercube_allreduce torus_2d 6 32 64 1024 1651.0399999999909 1753.2899999999754
28 intercube_allreduce torus_2d 6 64 128 2048 1660.0399999999881 1823.539999999979
29 intercube_allreduce torus_2d 6 128 256 4096 1678.0399999999827 1964.0399999999754
30 intercube_allreduce torus_2d 6 512 1024 16384 1795.0399999999827 2196.2849999999653
31 intercube_allreduce torus_2d 6 1024 2048 32768 1957.0399999999827 2476.74499999995
32 intercube_allreduce torus_2d 6 2048 4096 65536 2281.0399999999827 3037.664999999919
33 intercube_allreduce torus_2d 6 4096 8192 131072 2929.039999999979 4159.50500000003
34 intercube_allreduce torus_2d 6 8192 16384 262144 4225.039999999986 6403.185000000081
35 intercube_allreduce torus_2d 6 16384 32768 524288 6817.039999999943 10890.544999998769
36 intercube_allreduce torus_2d 6 32768 65536 1048576 12001.03999999992 19865.265000008738
37 intercube_allreduce torus_2d 6 49152 98304 1572864 17185.039999999994 28839.985000013185
Binary file not shown.

Before

Width:  |  Height:  |  Size: 38 KiB

After

Width:  |  Height:  |  Size: 36 KiB