From c7b1d802f113785b968dc2596dbf144eda1c9da3 Mon Sep 17 00:00:00 2001 From: qazal <77887910+Qazalin@users.noreply.github.com> Date: Sun, 26 May 2024 13:11:42 +0800 Subject: [PATCH] delete duplicate tests in test_linearizer (#4723) * delete duplicate test test_simplify_uop isnt needed max works * ci * remove skip * add skip back --- test/test_linearizer.py | 40 +++------------------------------------- test/test_uops.py | 5 +++-- 2 files changed, 6 insertions(+), 39 deletions(-) diff --git a/test/test_linearizer.py b/test/test_linearizer.py index e3a110afff..746795dc12 100644 --- a/test/test_linearizer.py +++ b/test/test_linearizer.py @@ -5,7 +5,6 @@ from dataclasses import replace from tinygrad.codegen.kernel import Opt, OptOps, KernelOptError from tinygrad.codegen.linearizer import Linearizer, UOp, UOps, expand_node, expand_idxs -from tinygrad.codegen.uops import UOpGraph from tinygrad.device import Device, Buffer from tinygrad.ops import BinaryOps, BufferOps, MemBuffer, ConstBuffer, LazyOp, LoadOps, TernaryOps, ReduceOps, UnaryOps from tinygrad.shape.shapetracker import ShapeTracker @@ -106,8 +105,8 @@ class TestLinearizer(unittest.TestCase): load_t = Tensor.full(load.st.shape, 1).contiguous().realize() k = helper_linearizer_ast(ast, [load_t], wanna_output=[load_t.numpy().sum()])[1] - self.assertEqual(k.uops.uops[-1].uop, UOps.ENDIF) - self.assertLess(k.uops.uops.index([x for x in k.uops.uops if x.uop is UOps.STORE][-1]), k.uops.uops.index(k.uops.uops[-1])) + self.assertEqual(k.uops[-1].uop, UOps.ENDIF) + self.assertLess(k.uops.uops.index([x for x in k.uops.uops if x.uop is UOps.STORE][-1]), k.uops.uops.index(k.uops[-1])) def test_two_nested_range(self): a = Tensor.randn(2, ).realize() @@ -496,16 +495,6 @@ class TestLinearizer(unittest.TestCase): tc_actions = [k for i, k in get_linearizer_actions(Linearizer(realized_ast), False).items() if k.applied_opts[0].op == OptOps.TC] assert len(tc_actions) == 9, f"get_linearizer_actions should contain 9 possible TC actions, only got {len(tc_actions)}" - @unittest.skipIf(Device.DEFAULT != "METAL", "these opts are only valid on METAL") - def test_tensor_cores_upcast_unroll(self): - ast = LazyOp(op=BufferOps.STORE, src=(LazyOp(op=BinaryOps.MUL, src=(LazyOp(op=BinaryOps.MUL, src=(LazyOp(op=UnaryOps.CAST, src=(LazyOp(op=ReduceOps.SUM, src=(LazyOp(op=UnaryOps.CAST, src=(LazyOp(op=BinaryOps.MUL, src=(LazyOp(op=BufferOps.LOAD, src=(), arg=MemBuffer(idx=1, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 4096), strides=(0, 4096, 0, 1), offset=0, mask=None, contiguous=False),)))), LazyOp(op=BufferOps.LOAD, src=(), arg=MemBuffer(idx=2, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 4096), strides=(0, 0, 4096, 1), offset=0, mask=None, contiguous=False),))))), arg=None),), arg=dtypes.float),), arg=(3,)),), arg=dtypes.half), LazyOp(op=BinaryOps.DIV, src=(LazyOp(op=BufferOps.CONST, src=(), arg=ConstBuffer(val=1.0, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 1), strides=(0, 0, 0, 0), offset=0, mask=None, contiguous=False),)))), LazyOp(op=BinaryOps.ADD, src=(LazyOp(op=BufferOps.CONST, src=(), arg=ConstBuffer(val=1.0, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 1), strides=(0, 0, 0, 0), offset=0, mask=None, contiguous=False),)))), LazyOp(op=UnaryOps.EXP2, src=(LazyOp(op=BinaryOps.MUL, src=(LazyOp(op=UnaryOps.CAST, src=(LazyOp(op=ReduceOps.SUM, src=(LazyOp(op=UnaryOps.CAST, src=(LazyOp(op=BinaryOps.MUL, src=(LazyOp(op=BufferOps.LOAD, src=(), arg=MemBuffer(idx=1, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 4096), strides=(0, 4096, 0, 1), offset=0, mask=None, contiguous=False),)))), LazyOp(op=BufferOps.LOAD, src=(), arg=MemBuffer(idx=2, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 4096), strides=(0, 0, 4096, 1), offset=0, mask=None, contiguous=False),))))), arg=None),), arg=dtypes.float),), arg=(3,)),), arg=dtypes.half), LazyOp(op=BufferOps.CONST, src=(), arg=ConstBuffer(val=-1.4426950408889634, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 1), strides=(0, 0, 0, 0), offset=0, mask=None, contiguous=False),))))), arg=None),), arg=None)), arg=None)), arg=None)), arg=None), LazyOp(op=UnaryOps.CAST, src=(LazyOp(op=BufferOps.LOAD, src=(), arg=MemBuffer(idx=3, dtype=dtypes.float, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 1), strides=(0, 11008, 1, 0), offset=0, mask=None, contiguous=True),)))),), arg=dtypes.half)), arg=None),), arg=MemBuffer(idx=0, dtype=dtypes.half, st=ShapeTracker(views=(View(shape=(1, 3, 11008, 1), strides=(0, 11008, 1, 0), offset=0, mask=None, contiguous=True),)))), # noqa: E501 - a = Tensor.empty(1, 3, 11008, 4096).realize() - b = Tensor.empty(1, 3, 11008, 4096).realize() - c = Tensor.empty(1, 3, 11008, 1).realize() - opt = [Opt(op=OptOps.TC, axis=0, amt=2), Opt(op=OptOps.LOCAL, axis=0, amt=4), Opt(op=OptOps.UNROLL, axis=0, amt=4), - Opt(op=OptOps.UPCAST, axis=5, amt=0)] - helper_linearizer_ast(ast, [a, b, c], opts=[opt]) - @unittest.skipIf(Device.DEFAULT != "METAL", "these opts are only valid on METAL") def test_tensor_cores_upcast_unroll_minimal(self): # the llama BEAM=2 failure is like this - float2 upcast of PHI should render inside the loop, cast_half should render outside the loop @@ -606,27 +595,6 @@ class TestLinearizer(unittest.TestCase): lin.linearize() assert not any(u.arg == TernaryOps.WHERE for u in lin.uops), "found where where where should be folded" - def test_simplify_uop(self): - def helper_test_simplify(uop, dtype, vin, arg=None): - ast = LazyOp(BufferOps.CONST, (), - ConstBuffer(42, dtypes.float, ShapeTracker(views=(View(shape=(), strides=(), offset=0, mask=None, contiguous=True),)))) - ast = LazyOp(BufferOps.STORE, (ast,), - MemBuffer(0, dtypes.float, ShapeTracker(views=(View(shape=(), strides=(), offset=0, mask=None, contiguous=True),)))) - lin = Linearizer(ast) # this is a dummy ast - - lin.uops = UOpGraph() - ret = lin.uops.add(uop, dtype, vin, arg) - lin.uops.add(UOps.SINK, None, (ret,)) - return list(lin.uops.uops)[-1] - - c0 = UOp(UOps.CONST, dtypes.float, vin=(), arg=0.0) - assert helper_test_simplify(UOps.ALU, dtypes.float, vin=(UOp(UOps.CONST, dtypes.bool, vin=(), arg=True), c0, c0), arg=TernaryOps.WHERE) == c0 - - c0 = UOp(UOps.CONST, dtypes.float, vin=(), arg=0.0) - c1 = UOp(UOps.CONST, dtypes.float, vin=(), arg=1.0) - assert helper_test_simplify(UOps.ALU, dtypes.float, vin=(UOp(UOps.CONST, dtypes.bool, vin=(), arg=True), c0, c1), - arg=TernaryOps.WHERE).arg == c0.arg - def test_phi_simplification(self): def helper(t, max_ops=0): sched = create_schedule([t.lazydata]) @@ -947,9 +915,7 @@ class TestHandCodedOpts(unittest.TestCase): b = Tensor.rand(N, N).realize() c = a @ b - s = create_schedule([c.lazydata])[0] - k = Linearizer(*s.ast) - k.hand_coded_optimizations() + k = helper_linearizer_opt(c)[-1] assert k.group_for_reduces == 1 assert k.local_dims == 1 diff --git a/test/test_uops.py b/test/test_uops.py index 8c918da085..ebd6375ce6 100644 --- a/test/test_uops.py +++ b/test/test_uops.py @@ -220,7 +220,8 @@ class TestConstantFolding(unittest.TestCase): class TestLocalAccess(unittest.TestCase): # NOTE: this is failing on METAL CI, no idea why. Works locally. - @unittest.skipIf(Device.DEFAULT in {"LLVM"} or (Device.DEFAULT == "METAL" and CI), "device doesn't support local memory") + @unittest.skipIf(Device.DEFAULT == "METAL" and CI, "failing only in CI") + @unittest.skipUnless(Device[Device.DEFAULT].renderer.has_local, "test requires locals") def test_local_basic(self): uops = [] smem = uop(uops, UOps.DEFINE_LOCAL, PtrDType(dtypes.float32), (), ('smem', 16)) @@ -229,7 +230,7 @@ class TestLocalAccess(unittest.TestCase): sres = uop(uops, UOps.LOAD, dtypes.float32, (smem, uop(uops, UOps.CONST, dtypes.int32, (), 0), barr)) self.assertEqual(_test_uops_result(dtypes.float32, uops, sres), 42) - @unittest.skipIf(Device.DEFAULT in {"LLVM"}, "device doesn't support local memory") + @unittest.skipUnless(Device[Device.DEFAULT].renderer.has_local, "test requires locals") def test_local_indirect(self): uops = [] smem = uop(uops, UOps.DEFINE_LOCAL, PtrDType(dtypes.int32), (), ('smem', 16))