hotfix: decrease dims in test_attention to get below the 90s limit
Unit Tests / Models (push) Successful in 1m43s
Unit Tests / Linters (push) Successful in 1m57s
Unit Tests / Linux (DSP) (push) Successful in 2m2s
Unit Tests / Test LLM (push) Failing after 2m29s
Unit Tests / Fuzzing (push) Successful in 2m30s
Unit Tests / Docs (push) Successful in 2m51s
Unit Tests / hcq2 (push) Successful in 3m0s
Unit Tests / AMD ASM IDE (push) Successful in 3m6s
Unit Tests / Python Backend (push) Successful in 3m15s
Unit Tests / openpilot Compile Tests (push) Successful in 3m19s
Unit Tests / Null Tests (push) Successful in 3m22s
Unit Tests / Unit Tests (push) Successful in 3m23s
Unit Tests / Torch Backend Training (push) Successful in 3m26s
Unit Tests / CL IMAGE Tests (push) Successful in 3m37s
Unit Tests / Linux (DEV=CPU:X86) (push) Successful in 3m36s
Unit Tests / SPEC=2 (2) (push) Successful in 3m58s
Unit Tests / Linux (amdllvm gfx1100) (push) Successful in 3m52s
Unit Tests / Linux (DEV=CPU:LVP) (push) Successful in 3m54s
Unit Tests / Linux (amdllvm gfx1201) (push) Successful in 4m0s
Unit Tests / SPEC=2 (1) (push) Successful in 4m9s
Unit Tests / Linux (DEV=CPU:LLVM) (push) Successful in 4m8s
Unit Tests / Optimization Tests (push) Successful in 4m18s
Unit Tests / Linux (am) (push) Successful in 4m16s
Unit Tests / Linux (DEV=CL) (push) Successful in 4m19s
Unit Tests / Torch Backend Tests (push) Successful in 4m25s
Unit Tests / Linux (amd gfx1100) (push) Successful in 4m19s
Unit Tests / ONNX (CPU) Tests (push) Failing after 4m26s
Unit Tests / Linux (amd gfx1201) (push) Successful in 4m22s
Unit Tests / Linux (DEV=WEBGPU) (push) Successful in 4m24s
Unit Tests / Compile-only (DEV=NULL:NAK:sm_120) (push) Successful in 1m44s
Deploy Docs / deploy (push) Successful in 4m52s
Unit Tests / Linux (DEV=CPU:CLANG) (push) Successful in 4m54s
Unit Tests / Compile-only (DEV=NULL:IR3:a630) (push) Successful in 2m31s
Unit Tests / Linux (amdllvm gfx950) (push) Successful in 3m31s
Unit Tests / Linux (ptx) (push) Successful in 3m25s
Unit Tests / Linux (nv) (push) Successful in 4m17s
Unit Tests / Linux (amd gfx950) (push) Successful in 4m59s
Unit Tests / Compile-only (DEV=NULL:QCOMCL:a630) (push) Successful in 4m29s
Autogen / In-tree Autogen (push) Successful in 12m31s
Autogen / In-tree Autogen (macos) (push) Canceled after 0s
Benchmarks / Mac pytest (push) Canceled after 0s
Benchmarks / LLM (DEV=AMD) (push) Canceled after 0s
Benchmarks / LLM (DEV=METAL) (push) Canceled after 0s
Benchmarks / LLM (DEV=NV) (push) Canceled after 0s
Benchmarks / HLB-CIFAR10 (DEV=AMD) (push) Canceled after 0s
Benchmarks / HLB-CIFAR10 (DEV=METAL) (push) Canceled after 0s
Benchmarks / HLB-CIFAR10 (DEV=NV) (push) Canceled after 0s
Benchmarks / MLPerf (AMD) (push) Canceled after 0s
Benchmarks / MLPerf (NV) (push) Canceled after 0s
Benchmarks / Stable Diffusion (DEV=AMD) (push) Canceled after 0s
Benchmarks / Stable Diffusion (DEV=METAL) (push) Canceled after 0s
Benchmarks / Stable Diffusion (DEV=NV) (push) Canceled after 0s
Benchmarks / Multi-GPU Benchmarks (DEV=AMD) (push) Canceled after 0s
Benchmarks / Multi-GPU Benchmarks (DEV=NV) (push) Canceled after 0s
Benchmarks / Tests (DEV=AMD) (push) Canceled after 0s
Benchmarks / Tests (DEV=METAL) (push) Canceled after 0s
Benchmarks / Tests (DEV=NV) (push) Canceled after 0s
Benchmarks / UsbGPU Benchmark (push) Canceled after 0s
Benchmarks / openpilot 0.11.0 compile3 dmonitoring (DEV=QCOM) (push) Canceled after 0s
Benchmarks / openpilot 0.11.2 compile3 dmonitoring (DEV=QCOM) (push) Canceled after 0s
Benchmarks / openpilot 0.11.0 compile3 policy (DEV=QCOM) (push) Canceled after 0s
Benchmarks / openpilot 0.11.2 compile3 supercombo (DEV=QCOM) (push) Canceled after 0s
Benchmarks / openpilot 0.11.0 compile3 vision (DEV=QCOM) (push) Canceled after 0s
Benchmarks / openpilot 0.11.0 compile3 dmonitoring (DEV=QCOM:IR3) (push) Canceled after 0s
Benchmarks / openpilot 0.11.2 compile3 dmonitoring (DEV=QCOM:IR3) (push) Canceled after 0s
Benchmarks / openpilot 0.11.0 compile3 policy (DEV=QCOM:IR3) (push) Canceled after 0s
Benchmarks / openpilot 0.11.2 compile3 supercombo (DEV=QCOM:IR3) (push) Canceled after 0s
Benchmarks / openpilot 0.11.0 compile3 vision (DEV=QCOM:IR3) (push) Canceled after 0s
Benchmarks / DSP Benchmark (push) Canceled after 0s
Benchmarks / UsbGPU Benchmark (comma) (push) Canceled after 0s
Benchmarks / PCI Driver Benchmark (DEV=AMD) (push) Canceled after 0s
Benchmarks / PCI Driver Benchmark (DEV=NV) (push) Canceled after 0s
Benchmarks / LLVM Speed (push) Canceled after 0s
Platform Tests / MacOS (unit) (push) Canceled after 0s
Platform Tests / MacOS (unit, mock) (push) Canceled after 0s
Platform Tests / MacOS (DEV=METAL) (1) (push) Canceled after 0s
Platform Tests / MacOS (DEV=METAL) (2) (push) Canceled after 0s
Platform Tests / MacOS (DEV=CPU:CLANG) (push) Canceled after 0s
Platform Tests / MacOS (DEV=CPU:LLVM) (push) Canceled after 0s
Platform Tests / MacOS (DEV=CPU:LVP) (push) Canceled after 0s
Platform Tests / MacOS (DEV=WEBGPU) (push) Canceled after 0s
Platform Tests / Windows (DEV=CPU:CLANG) (push) Canceled after 0s
Platform Tests / Windows (DEV=CPU:LLVM) (push) Canceled after 0s
Platform Tests / Windows (DEV=CPU:X86) (push) Canceled after 0s
Platform Tests / Windows (DEV=WEBGPU) (push) Canceled after 0s

This commit is contained in:
2026-08-27 14:44:06 -07:00
parent 38fa0643cf
commit ee3161e924
+8 -7
View File
@@ -73,10 +73,10 @@ class TestGatedDeltaNetBlock(unittest.TestCase):
return Tensor.linspace(start, stop, int(np.prod(shape)), dtype=dtypes.float32).reshape(*shape)
def _make_config(self, **kwargs):
return TransformerConfig(**({"num_blocks":1, "dim":32, "hidden_dim":64, "n_heads":1, "n_kv_heads":1,
"norm_eps":1e-5, "vocab_size":32, "head_dim":32, "rope_theta":10000.0,
"rope_dim":32, "v_head_dim":32, "max_context":4, "ssm_layers":(True,),
"ssm":SSMConfig(conv_kernel=2, state_size=32, group_count=1, time_step_rank=1, inner_size=32)} | kwargs))
return TransformerConfig(**({"num_blocks":1, "dim":8, "hidden_dim":16, "n_heads":1, "n_kv_heads":1,
"norm_eps":1e-5, "vocab_size":32, "head_dim":8, "rope_theta":10000.0,
"rope_dim":8, "v_head_dim":8, "max_context":4, "ssm_layers":(True,),
"ssm":SSMConfig(conv_kernel=2, state_size=4, group_count=1, time_step_rank=1, inner_size=4)} | kwargs))
def _make_block(self, config:TransformerConfig) -> GatedDeltaNetBlock:
block = GatedDeltaNetBlock(config, config.ssm)
@@ -229,7 +229,7 @@ class TestGatedDeltaNetBlock(unittest.TestCase):
np.testing.assert_allclose(block.recurrent_state.numpy(), initial_state.numpy() * alpha[..., None], rtol=1e-5, atol=1e-5)
def test_kda_prefill_matches_decode(self):
config = self._make_config(ssm=SSMConfig(conv_kernel=2, state_size=32, group_count=1, time_step_rank=1, inner_size=32, kda=True))
config = self._make_config(ssm=SSMConfig(conv_kernel=2, state_size=4, group_count=1, time_step_rank=1, inner_size=4, kda=True))
block = GatedDeltaNetBlock(config, config.ssm)
for p in nn.state.get_parameters(block):
p.replace(self._tensor_linspace(-0.05, 0.05, p.shape) if len(p.shape) > 1 else self._tensor_linspace(0.05, 0.1, p.shape))
@@ -245,7 +245,7 @@ class TestGatedDeltaNetBlock(unittest.TestCase):
def test_varied_chunk_sizes_match_decode(self):
for kda in (False, True):
ssm = SSMConfig(conv_kernel=2, state_size=32, group_count=1, time_step_rank=1, inner_size=32, kda=kda)
ssm = SSMConfig(conv_kernel=2, state_size=4, group_count=1, time_step_rank=1, inner_size=4, kda=kda)
config = self._make_config(ssm=ssm)
if kda:
block = GatedDeltaNetBlock(config, config.ssm)
@@ -267,7 +267,8 @@ class TestGatedDeltaNetBlock(unittest.TestCase):
np.testing.assert_allclose(chunked_recurrent, decode_recurrent, rtol=1e-3, atol=1e-3, err_msg=f"{kda=} {chunking=}")
def test_start_zero_resets_realized_state(self):
config, x = self._make_config(max_context=3), self._tensor_linspace(-1, 1, (1, 3, 32))
config = self._make_config(max_context=3)
x = self._tensor_linspace(-1, 1, (1, 3, config.dim))
block = self._make_block(config)
self._run_attention(block, x, 0)
restarted = self._run_attention(block, x[:, :2], 0)