From 4257939e50739280203b0aec191efbcb7101a670 Mon Sep 17 00:00:00 2001 From: nimlgen <138685161+nimlgen@users.noreply.github.com> Date: Tue, 14 Jul 2026 19:47:22 +0300 Subject: [PATCH] remove copyin/copyout from Buffer (#17020) * remove copyin/copyout from Buffer * x * x * x * x --- extra/hcq2/hcq2.py | 19 ----------------- test/backend/test_graph.py | 11 ++++------ test/backend/test_linearizer.py | 2 +- test/backend/test_profiler.py | 16 +++++++------- test/backend/test_renderer_failures.py | 6 +++--- test/backend/test_subbuffer.py | 23 ++++++++++---------- test/backend/test_uops.py | 14 ++++--------- test/device/test_hcq.py | 4 ++-- test/device/test_ocl.py | 4 ++-- test/external/external_test_hcq.py | 4 ++-- test/external/fuzz_graph.py | 10 ++++----- test/opt/test_tensor_cores.py | 3 ++- test/unit/test_invalid_tensor.py | 3 ++- tinygrad/device.py | 29 +++++++++++--------------- tinygrad/engine/realize.py | 5 ++--- tinygrad/runtime/ops_python.py | 9 +++++--- tinygrad/runtime/support/hcq.py | 5 +++-- tinygrad/tensor.py | 2 +- tinygrad/uop/ops.py | 3 +-- 19 files changed, 70 insertions(+), 102 deletions(-) diff --git a/extra/hcq2/hcq2.py b/extra/hcq2/hcq2.py index 01d9901de5..1dc882f874 100644 --- a/extra/hcq2/hcq2.py +++ b/extra/hcq2/hcq2.py @@ -593,22 +593,3 @@ class HCQAllocator(LRUAllocator[HCQDeviceType], Generic[HCQDeviceType]): self.dev.iface.free(mb) def _offset(self, buf, size:int, offset:int) -> HCQ2Buffer: return buf.offset(offset=offset, size=size) - - def _wrap(self, dev:str, sz:int, opaque:HCQ2Buffer) -> Buffer: - return Buffer(dev, sz, dtypes.uint8, opaque=opaque, options=BufferSpec(external_ptr=1)) - - def _copy(self, dst:Buffer, src:Buffer): - from tinygrad.engine.realize import run_linear - du, su = UOp.from_buffer(dst), UOp.from_buffer(src) - run_linear(UOp(Ops.LINEAR, src=(su.param_like(1).copy_to_device(dst.device).call(du, su),)), update_stats=True) - - def _copyin(self, dest:HCQ2Buffer, src:memoryview): - s = Buffer(self.dev.device, len(src), dtypes.uint8, options=BufferSpec(host=True), preallocate=True) - s._buf.cpu_view()[:len(src)] = src - self._copy(self._wrap(self.dev.device, len(src), dest), s) - - def _copyout(self, dest:memoryview, src:HCQ2Buffer): - d = Buffer(self.dev.device, len(dest), dtypes.uint8, options=BufferSpec(host=True), preallocate=True) - self._copy(d, self._wrap(self.dev.device, len(dest), src)) - self.dev.synchronize() - dest[:] = d._buf.cpu_view()[:len(dest)] diff --git a/test/backend/test_graph.py b/test/backend/test_graph.py index 20fefe99c3..6d16dd5f2d 100644 --- a/test/backend/test_graph.py +++ b/test/backend/test_graph.py @@ -1,9 +1,9 @@ import numpy as np -import functools, unittest, ctypes +import functools, unittest from tinygrad.device import Device, Buffer from tinygrad.tensor import Tensor -from tinygrad.helpers import Context, from_mv +from tinygrad.helpers import Context from tinygrad.dtype import dtypes from tinygrad.engine.jit import MultiGraphRunner from tinygrad.engine.realize import run_linear, compile_linear @@ -31,7 +31,7 @@ def make_buffer(device, size=BUF_SIZE, fill=False): buf = Buffer(device, size, dtypes.int).ensure_allocated() if fill: with Context(DEBUG=0): - buf.copyin(Tensor(np.random.randint(-10000, 10000, size=size, dtype=np.int32)).realize().uop.base.realized.as_memoryview()) + buf.copy_from(Tensor(np.random.randint(-10000, 10000, size=size, dtype=np.int32)).realize().uop.base.realized) return buf def make_view(base, offset_elems, size_elems): @@ -55,10 +55,7 @@ def run_schedule(calls:list[UOp]): run_linear(UOp(Ops.LINEAR, src=tuple(calls))) def zero_bufs(bufs): - for b in bufs: - mv = memoryview(bytearray(b.nbytes)) - ctypes.memset(from_mv(mv), 0, len(mv)) - b.copyin(mv) + for b in bufs: b.copy_from(Buffer("PYTHON", b.size, b.dtype, opaque=memoryview(bytearray(b.nbytes)))) @unittest.skipUnless(Device[Device.DEFAULT].graph is not None, "graph support required") class TestGraph(unittest.TestCase): diff --git a/test/backend/test_linearizer.py b/test/backend/test_linearizer.py index cac2bd0ad0..cca3c7c7f8 100644 --- a/test/backend/test_linearizer.py +++ b/test/backend/test_linearizer.py @@ -424,7 +424,7 @@ def copyout_outputs(outbufs:list[Buffer]) -> list[np.ndarray]: return [np.frombuffer(x.as_memoryview(), _to_np_dtype(x.dtype)) for x in outbufs] def reset_bufs(bufs:list[Buffer]): - for buf in bufs: buf.copyin(np.zeros((buf.size*buf.dtype.itemsize,), dtype=np.uint8).data) + for buf in bufs: buf.copy_from(Buffer("PYTHON", buf.size, buf.dtype, opaque=memoryview(bytearray(buf.nbytes)))) def _helper_linearizer_opt_ast(realized_ast:UOp, real_bufs:list[Buffer], opts=[], apply_tc=False, atol=1e-4, rtol=1e-4, color_sizes=[], wanna_output=[]): diff --git a/test/backend/test_profiler.py b/test/backend/test_profiler.py index 2091f85ac4..7c2796f7e1 100644 --- a/test/backend/test_profiler.py +++ b/test/backend/test_profiler.py @@ -70,9 +70,9 @@ class TestProfiler(unittest.TestCase): buf1 = Buffer(Device.DEFAULT, 2, dtypes.float, options=BufferSpec(nolru=True)).ensure_allocated() with helper_collect_profile(TestProfiler.d0) as profile: - buf1.copyin(memoryview(bytearray(struct.pack("ff", 0, 1)))) + buf1.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 1))))) - kernel_runs = [x for x in profile if isinstance(x, ProfileRangeEvent) and x.device.startswith(TestProfiler.d0.device)] + kernel_runs = [x for x in profile if isinstance(x, ProfileRangeEvent) and x.device.startswith((TestProfiler.d0.device, "PYTHON"))] assert len(kernel_runs) == 1, "one kernel run is expected" def test_profile_multiops(self): @@ -80,12 +80,12 @@ class TestProfiler(unittest.TestCase): buf1 = Buffer(Device.DEFAULT, 2, dtypes.float, options=BufferSpec(nolru=True)).ensure_allocated() with helper_collect_profile(TestProfiler.d0) as profile: - buf1.copyin(memoryview(bytearray(struct.pack("ff", 0, 1)))) + buf1.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 1))))) gs, ls = TestProfiler.prg.arg.launch_dims({}) TestProfiler.runtime(buf1._buf, TestProfiler.a.uop.buffer._buf, global_size=gs, local_size=ls) - buf1.copyout(memoryview(bytearray(buf1.nbytes))) + buf1.as_memoryview() - evs = [x for x in profile if isinstance(x, ProfileRangeEvent) and x.device.startswith(TestProfiler.d0.device)] + evs = [x for x in profile if isinstance(x, ProfileRangeEvent) and x.device.startswith((TestProfiler.d0.device, "PYTHON"))] assert len(evs) == 3, "3 kernel runs are expected" # NOTE: order of events does not matter, the tool is responsible for sorting them @@ -103,12 +103,12 @@ class TestProfiler(unittest.TestCase): buf2 = Buffer(f"{Device.DEFAULT}:1", 2, dtypes.float, options=BufferSpec(nolru=True)).ensure_allocated() with helper_collect_profile(TestProfiler.d0, d1) as profile: - buf1.copyin(memoryview(bytearray(struct.pack("ff", 0, 1)))) - buf2.copyin(memoryview(bytearray(struct.pack("ff", 0, 1)))) + buf1.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 1))))) + buf2.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 1))))) for dev in [TestProfiler.d0.device, d1.device]: evs = [x for x in profile if isinstance(x, ProfileRangeEvent) and _dev_base(x.device) == dev] - assert len(evs) == 1, "one kernel runs are expected" + assert len(evs) == (0 if hasattr(TestProfiler.d0.allocator, '_as_buffer') else 1), "one kernel runs are expected" def test_profile_multidev_transfer(self): try: d1 = Device[f"{Device.DEFAULT}:1"] diff --git a/test/backend/test_renderer_failures.py b/test/backend/test_renderer_failures.py index 867980036b..b347dcc8d1 100644 --- a/test/backend/test_renderer_failures.py +++ b/test/backend/test_renderer_failures.py @@ -1,6 +1,6 @@ import unittest import numpy as np -from tinygrad.device import Device +from tinygrad.device import Device, Buffer from tinygrad.dtype import dtypes, ConstType from tinygrad.engine.realize import run_linear from tinygrad.codegen import to_program @@ -10,13 +10,13 @@ from tinygrad.renderer.ptx import PTXRenderer from tinygrad.renderer.wgsl import WGSLRenderer from tinygrad.runtime.ops_python import PythonRenderer from tinygrad.uop.ops import UOp, Ops, KernelInfo, python_alu -from tinygrad.tensor import Tensor, _to_np_dtype +from tinygrad.tensor import Tensor def _test_uop_result(inputs:list[Tensor], sink:UOp, local_size=None): for x in inputs: x.realize() sz = 1 if local_size is None else prod(local_size) outs = [UOp.new_buffer(Device.DEFAULT, sz, u.src[1].dtype) for u in sink.src if u.op is Ops.STORE] - for u in outs: u.buffer.allocate().copyin(np.zeros(sz, dtype=_to_np_dtype(u.dtype)).data) + for u in outs: u.buffer.allocate().copy_from(Buffer("PYTHON", sz, u.dtype, opaque=memoryview(bytearray(u.buffer.nbytes)))) run_linear(UOp(Ops.LINEAR, src=(sink.call(*outs, *(x.uop.base for x in inputs)),))) return [u.buffer.numpy() for u in outs] diff --git a/test/backend/test_subbuffer.py b/test/backend/test_subbuffer.py index 8b1b3aef3c..69fdafdab1 100644 --- a/test/backend/test_subbuffer.py +++ b/test/backend/test_subbuffer.py @@ -7,8 +7,7 @@ from test.helpers import needs_second_gpu @unittest.skipIf(Device.DEFAULT in {"WEBGPU", "CL"}, "subbuffer not supported") class TestSubBuffer(unittest.TestCase): def setUp(self): - self.buf = Buffer(Device.DEFAULT, 10, dtypes.uint8).ensure_allocated() - self.buf.copyin(memoryview(bytearray(range(10)))) + self.buf = Buffer(Device.DEFAULT, 10, dtypes.uint8, initial_value=bytes(range(10))) self.buf_unalloc = Buffer(Device.DEFAULT, 10, dtypes.uint8) def test_subbuffer(self): @@ -59,7 +58,7 @@ class TestSubBuffer(unittest.TestCase): _ = Buffer(Device.DEFAULT, 10, dtypes.uint8).ensure_allocated() self.buf.ensure_allocated() - self.buf.copyin(memoryview(bytearray(range(10, 20)))) + self.buf.copy_from(Buffer("PYTHON", 10, dtypes.uint8, opaque=memoryview(bytearray(range(10, 20))))) vbuf.ensure_allocated() @@ -109,15 +108,15 @@ class TestSubBuffer(unittest.TestCase): def test_subbuffer_copy_in_out(self): sub_buf = self.buf.view(3, dtypes.uint8, offset=3).ensure_allocated() # [3:6] data_out_sub = bytearray([0]*3) - sub_buf.copyout(memoryview(data_out_sub)) + data_out_sub[:] = sub_buf.as_memoryview() assert data_out_sub == bytearray(range(3, 6)) - sub_buf.copyin(memoryview(bytearray(range(3)))) + sub_buf.copy_from(Buffer("PYTHON", 3, dtypes.uint8, opaque=memoryview(bytearray(range(3))))) assert sub_buf.as_memoryview().tolist() == list(range(3)) assert self.buf.as_memoryview().tolist()[3:6] == list(range(3)) - sub_buf.copyout(memoryview(data_out_sub)) + data_out_sub[:] = sub_buf.as_memoryview() assert data_out_sub == bytearray(range(3)) data_out_base = bytearray([0]*10) - self.buf.copyout(memoryview(data_out_base)) + data_out_base[:] = self.buf.as_memoryview() assert data_out_base[0:3] == bytearray(range(0, 3)) assert data_out_base[3:6] == data_out_sub assert data_out_base[6:10] == bytearray(range(6, 10)) @@ -129,27 +128,27 @@ class TestSubBuffer(unittest.TestCase): self.assertTrue(view2.is_allocated()) data_in = bytearray([7, 8, 9]) - view2.copyin(memoryview(data_in)) + view2.copy_from(Buffer("PYTHON", 3, view2.dtype, opaque=memoryview(data_in))) data_out_v2 = bytearray([0]*3) - view2.copyout(memoryview(data_out_v2)) + data_out_v2[:] = view2.as_memoryview() assert data_in == data_out_v2 expected_base_data = memoryview(bytearray(range(10))) expected_base_data[4:7] = data_in data_out_base = bytearray([0]*10) - self.buf.copyout(memoryview(data_out_base)) + data_out_base[:] = self.buf.as_memoryview() assert expected_base_data == data_out_base def test_subbuffer_alloc(self): sub_buf = self.buf.view(4, dtypes.int8, offset=3) sub_buf.allocate() - sub_buf.copyin(memoryview(bytearray(range(10, 14)))) + sub_buf.copy_from(Buffer("PYTHON", 4, dtypes.int8, opaque=memoryview(bytearray(range(10, 14))))) assert self.buf.as_memoryview().tolist()[3:7] == sub_buf.as_memoryview().tolist() sub_buf = self.buf_unalloc.view(4, dtypes.int8, offset=3) sub_buf.allocate() - sub_buf.copyin(memoryview(bytearray(range(10, 14)))) + sub_buf.copy_from(Buffer("PYTHON", 4, dtypes.int8, opaque=memoryview(bytearray(range(10, 14))))) assert self.buf_unalloc.as_memoryview().tolist()[3:7] == sub_buf.as_memoryview().tolist() def test_subbuffer_dealloc(self): diff --git a/test/backend/test_uops.py b/test/backend/test_uops.py index 31a7647a51..af1164fb1b 100644 --- a/test/backend/test_uops.py +++ b/test/backend/test_uops.py @@ -33,11 +33,9 @@ def _test_single_value(vals, op, dts): alu = uop(uops, op, output_dtype, loads) out = uop(uops, Ops.STORE, dtypes.void, (buf_store.index(uop(uops, Ops.CONST, dtypes.int32, (), 0)), alu)) buf = Buffer(Device.DEFAULT, 1, output_dtype).allocate() - buf2 = [Buffer(Device.DEFAULT, 1, dtype).allocate().copyin(np.array([a], dtype=_to_np_dtype(dtype)).data) for a,dtype in zip(vals, dts)] + buf2 = [Buffer(Device.DEFAULT, 1, dtype, initial_value=np.array([a], dtype=_to_np_dtype(dtype)).tobytes()) for a,dtype in zip(vals, dts)] run_uops([out], [buf]+buf2) - ret = np.empty(1, _to_np_dtype(output_dtype)) - buf.copyout(ret.data) - return ret[0] + return np.frombuffer(buf.as_memoryview(), _to_np_dtype(output_dtype))[0] def _test_single_value_const(vals, op, dts): uops = [] @@ -48,9 +46,7 @@ def _test_single_value_const(vals, op, dts): out = buf_store[UOp.const(dtypes.int32, 0)].store(alu) buf = Buffer(Device.DEFAULT, 1, output_dtype).allocate() run_uops([out], [buf]) - ret = np.empty(1, _to_np_dtype(output_dtype)) - buf.copyout(ret.data) - return ret[0] + return np.frombuffer(buf.as_memoryview(), _to_np_dtype(output_dtype))[0] def _test_uops_result(output_dtype, uops, res): # uops = [] @@ -59,9 +55,7 @@ def _test_uops_result(output_dtype, uops, res): out = uop(uops, Ops.STORE, dtypes.void, (buf_store.index(uop(uops, Ops.CONST, dtypes.int32, (), 0)), res)) buf = Buffer(Device.DEFAULT, 1, output_dtype).allocate() run_uops([out], [buf]) - ret = np.empty(1, _to_np_dtype(output_dtype)) - buf.copyout(ret.data) - return ret[0] + return np.frombuffer(buf.as_memoryview(), _to_np_dtype(output_dtype))[0] class TestUOps(unittest.TestCase): def _equal(self, v1, v2): diff --git a/test/device/test_hcq.py b/test/device/test_hcq.py index e0d4fe32e4..369adddf8d 100644 --- a/test/device/test_hcq.py +++ b/test/device/test_hcq.py @@ -31,8 +31,8 @@ class TestHCQ(unittest.TestCase): def setUp(self): TestHCQ.d0.synchronize() - TestHCQ.a.uop.buffer.copyin(memoryview(bytearray(struct.pack("ff", 0, 1)))) - TestHCQ.b.uop.buffer.copyin(memoryview(bytearray(struct.pack("ff", 0, 0)))) + TestHCQ.a.uop.buffer.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 1))))) + TestHCQ.b.uop.buffer.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 0))))) TestHCQ.d0.synchronize() # wait for copyins to complete # Test signals diff --git a/test/device/test_ocl.py b/test/device/test_ocl.py index 3923a95093..90bfbe55e4 100644 --- a/test/device/test_ocl.py +++ b/test/device/test_ocl.py @@ -34,7 +34,7 @@ class TestCLError(unittest.TestCase): data = list(range(65)) unaligned = memoryview(bytearray(data))[1:] buffer = Buffer("CL", 64, dtypes.uint8).allocate() - buffer.copyin(unaligned) + buffer.copy_from(Buffer("PYTHON", 64, dtypes.uint8, opaque=unaligned)) result = memoryview(bytearray(len(data) - 1)) - buffer.copyout(result) + result[:] = buffer.as_memoryview() assert unaligned == result, "Unaligned data copied in must be equal to data copied out." diff --git a/test/external/external_test_hcq.py b/test/external/external_test_hcq.py index cb21563ae1..ec65fffd7e 100644 --- a/test/external/external_test_hcq.py +++ b/test/external/external_test_hcq.py @@ -47,8 +47,8 @@ class TestHCQ(unittest.TestCase): def setUp(self): TestHCQ.d0.synchronize() - TestHCQ.a.uop.buffer.copyin(memoryview(bytearray(struct.pack("ff", 0, 1)))) - TestHCQ.b.uop.buffer.copyin(memoryview(bytearray(struct.pack("ff", 0, 0)))) + TestHCQ.a.uop.buffer.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 1))))) + TestHCQ.b.uop.buffer.copy_from(Buffer("PYTHON", 2, dtypes.float, opaque=memoryview(bytearray(struct.pack("ff", 0, 0))))) TestHCQ.d0.synchronize() # wait for copyins to complete def test_run_1000_times_one_submit(self): diff --git a/test/external/fuzz_graph.py b/test/external/fuzz_graph.py index edc2c6faee..e2a6877298 100644 --- a/test/external/fuzz_graph.py +++ b/test/external/fuzz_graph.py @@ -1,7 +1,7 @@ -import random, ctypes +import random import numpy as np from tinygrad.device import Buffer, Device -from tinygrad.helpers import Context, getenv, from_mv +from tinygrad.helpers import Context, getenv from tinygrad.dtype import dtypes from tinygrad.tensor import Tensor, _to_np_dtype from tinygrad.engine.realize import BufferXfer, get_runner, ExecItem @@ -29,7 +29,7 @@ def alloc_rawbuffer(device, fill=False): if fill: with Context(DEBUG=0): data = np.random.randint(-10000, 10000, size=rawbuf.size, dtype=_to_np_dtype(rawbuf.dtype)) - rawbuf.copyin(Tensor(data).realize().uop.base.realized.as_memoryview()) + rawbuf.copy_from(Tensor(data).realize().uop.base.realized) return rawbuf def gen_kernel_ji(device, deps): @@ -84,9 +84,7 @@ def run_jit(jis, all_buffers, input_buffers, var_vals): with Context(DEBUG=0): for rawbuf in all_buffers: if rawbuf in input_buffers: continue - mv = memoryview(bytearray(rawbuf.nbytes)) - ctypes.memset(from_mv(mv), 0, len(mv)) - rawbuf.copyin(mv) + rawbuf.copy_from(Buffer("PYTHON", rawbuf.size, rawbuf.dtype, opaque=memoryview(bytearray(rawbuf.nbytes)))) for ei in jis: ei.run(var_vals, jit=True) diff --git a/test/opt/test_tensor_cores.py b/test/opt/test_tensor_cores.py index 9c269f1f6d..d9595fe0b8 100644 --- a/test/opt/test_tensor_cores.py +++ b/test/opt/test_tensor_cores.py @@ -149,7 +149,8 @@ class TestTensorCores(unittest.TestCase): # TODO: support this even if numpy doesn't if _to_np_dtype(real_bufs[0].dtype) is None: continue - real_bufs[0].copyin(np.zeros((real_bufs[0].size, ), dtype=_to_np_dtype(real_bufs[0].dtype)).data) # Zero to check that all values are filled + # Zero to check that all values are filled + real_bufs[0].copy_from(Buffer("PYTHON", real_bufs[0].size, real_bufs[0].dtype, opaque=memoryview(bytearray(real_bufs[0].nbytes)))) run_program(ast, real_bufs) result = np.frombuffer(real_bufs[0].as_memoryview(), _to_np_dtype(real_bufs[0].dtype)) diff --git a/test/unit/test_invalid_tensor.py b/test/unit/test_invalid_tensor.py index 2e051cbac9..14b6a8570b 100644 --- a/test/unit/test_invalid_tensor.py +++ b/test/unit/test_invalid_tensor.py @@ -1,5 +1,6 @@ import unittest from tinygrad import Tensor +from tinygrad.device import Buffer from tinygrad.dtype import Invalid, dtypes from tinygrad.engine.realize import run_linear @@ -9,7 +10,7 @@ class TestInvalidTensor(unittest.TestCase): buf = out.uop.buffer buf.allocate() sentinel = memoryview(bytearray(b'\x42' * buf.nbytes)) - buf.copyin(sentinel) + buf.copy_from(Buffer("PYTHON", buf.size, buf.dtype, opaque=sentinel)) before = buf.as_memoryview().cast(out.dtype.fmt).tolist() run_linear(linear, var_vals) ret = buf.as_memoryview().cast(out.dtype.fmt).tolist() diff --git a/tinygrad/device.py b/tinygrad/device.py index b53e65faa1..f3cdd0855d 100644 --- a/tinygrad/device.py +++ b/tinygrad/device.py @@ -3,7 +3,7 @@ from dataclasses import dataclass, replace from collections import defaultdict from typing import Any, Generic, TypeVar, Iterator, Generator, TYPE_CHECKING import importlib, inspect, functools, pathlib, os, contextlib, re, atexit, pickle, decimal -from tinygrad.helpers import LRU, getenv, diskcache_get, diskcache_put, DEBUG, GlobalCounters, flat_mv, PROFILE, temp, colored +from tinygrad.helpers import LRU, getenv, diskcache_get, diskcache_put, DEBUG, GlobalCounters, PROFILE, temp, colored from tinygrad.helpers import Context, CCACHE, ALLOW_DEVICE_USAGE, MAX_BUFFER_SIZE, cpu_events, ProfileEvent, ProfilePointEvent, suppress_finalizing from tinygrad.helpers import select_by_name, select_first_inited, DEV, TracingKey, size_to_str, pluralize from tinygrad.dtype import DType, _to_np_dtype @@ -112,7 +112,7 @@ class Buffer: if opaque is not None: self.allocate(opaque) if initial_value is not None: self.allocate() - self.copyin(memoryview(initial_value)) + self.copy_from(Buffer("PYTHON", self.size, self.dtype, opaque=memoryview(bytearray(initial_value)))) else: assert base._base is None, "base can't have a base" assert device == base.device, "base must have the same device" @@ -176,9 +176,7 @@ class Buffer: if self._base is not None: return self.__class__, (self.device, self.size, self.dtype, None, None, None, 0, self.base, self.offset, self.is_allocated()) if self.device == "NPY": return self.__class__, (self.device, self.size, self.dtype, self._buf, self.options, None, self.uop_refcount) - if self.is_allocated(): - buf = bytearray(self.nbytes) - self.copyout(memoryview(buf)) + if self.is_allocated(): buf = bytearray(self.as_memoryview()) return self.__class__, (self.device, self.size, self.dtype, None, self.options, buf, self.uop_refcount) @property def trace_num(self) -> int: @@ -195,23 +193,20 @@ class Buffer: # zero copy with as_memoryview (disabled by default due to use after free) if (force_zero_copy or allow_zero_copy) and hasattr(self.allocator, '_as_buffer'): return self.allocator._as_buffer(self._buf) assert not force_zero_copy, "force zero copy was passed, but copy is required" - return self.copyout(memoryview(bytearray(self.nbytes))) + Buffer("PYTHON", self.size, self.dtype, opaque=(mv:=memoryview(bytearray(self.nbytes)))).copy_from(self) + return mv def numpy(self) -> 'np.ndarray': # type: ignore [name-defined] # noqa: F821 import numpy as np assert _to_np_dtype(self.dtype) is not None, f"no np dtype for {self.dtype}" return np.frombuffer(self.as_memoryview(), dtype=_to_np_dtype(self.dtype)) - def copyin(self, mv:memoryview): - mv = flat_mv(mv) - assert len(mv) == self.nbytes, f"size mismatch, {len(mv)=} != {self.dtype=} {self.size=}" - assert self.is_initialized(), "can't copyin to unallocated buffer" - self.allocator._copyin(self._buf, mv) + def copy_from(self, src:Buffer) -> Buffer: + assert self.nbytes == src.nbytes, f"copy size mismatch, {self.nbytes} != {src.nbytes}" + assert self.is_initialized() and src.is_initialized(), "copy requires allocated buffers" + from tinygrad.engine.realize import run_linear + from tinygrad.uop.ops import UOp, Ops + du, su = UOp.from_buffer(self), UOp.from_buffer(src) + run_linear(UOp(Ops.LINEAR, src=(su.param_like(1).copy_to_device(self.device).call(du, su),)), update_stats=False) return self - def copyout(self, mv:memoryview) -> memoryview: - mv = flat_mv(mv) - assert len(mv) == self.nbytes, f"size mismatch, {len(mv)=} != {self.dtype=} {self.size=}" - assert self.is_initialized(), "can't copyout unallocated buffer" - self.allocator._copyout(mv, self._buf) - return mv def view(self, size:int, dtype:DType, offset:int) -> Buffer: assert offset < self.nbytes, "offset must be less than nbytes" return Buffer(self.device, size, dtype, base=self.base, offset=self.offset+offset) diff --git a/tinygrad/engine/realize.py b/tinygrad/engine/realize.py index 0b68e4100d..3a64d7b0bf 100644 --- a/tinygrad/engine/realize.py +++ b/tinygrad/engine/realize.py @@ -166,9 +166,8 @@ def exec_copy(ctx:ExecContext, call:UOp, ast:UOp) -> float|None: elif src.device.startswith("DISK") and getattr(src.allocator.dev, 'fd', None) is not None \ and hasattr(dest.allocator, 'copy_from_disk') and src.nbytes >= 4096 and dest.allocator.supports_copy_from_disk: dest.allocator.copy_from_disk(dest._buf, src._buf, src.nbytes) - elif src.device.startswith(("DISK", "TINYFS")) and hasattr(dest.allocator, '_as_buffer'): - src.allocator._copyout(dest.allocator._as_buffer(dest._buf), src._buf) - else: dest.copyin(src.as_memoryview(allow_zero_copy=True)) + elif hasattr(dest.allocator, '_as_buffer'): src.allocator._copyout(dest.allocator._as_buffer(dest._buf), src._buf) + else: dest.allocator._copyin(dest._buf, src.as_memoryview(allow_zero_copy=True)) return None def exec_kernel(ctx:ExecContext, call:UOp, ast:UOp) -> float|None: diff --git a/tinygrad/runtime/ops_python.py b/tinygrad/runtime/ops_python.py index 766ae734d1..44def8a56d 100644 --- a/tinygrad/runtime/ops_python.py +++ b/tinygrad/runtime/ops_python.py @@ -6,7 +6,7 @@ from typing import Any, TYPE_CHECKING import pickle, base64, itertools, time, sys, functools from dataclasses import replace from tinygrad.dtype import DType, dtypes, AddrSpace, truncate, storage_fmt_for_dtype, to_storage_scalar, from_storage_scalar -from tinygrad.helpers import all_same, getenv, flatten, Target, IMAGE, is_image_shape +from tinygrad.helpers import all_same, getenv, flatten, Target, IMAGE, is_image_shape, cpu_profile from tinygrad.device import Buffer, Compiled, Compiler, Allocator from tinygrad.codegen.opt import tc from tinygrad.uop.ops import exec_alu, python_alu, Ops, UOp, GroupOp, bitcast @@ -223,8 +223,11 @@ class PythonRenderer(Renderer): class PythonAllocator(Allocator['PythonDevice']): def _alloc(self, size, options): return memoryview(bytearray(size)) - def _copyin(self, dest, src:memoryview): dest[:] = src - def _copyout(self, dest:memoryview, src): dest[:] = src + def _as_buffer(self, src) -> memoryview: return src + def _copyin(self, dest, src:memoryview): + with cpu_profile("TINY -> PYTHON", f"{self.dev.device}:COPY"): dest[:] = src + def _copyout(self, dest:memoryview, src): + with cpu_profile("PYTHON -> TINY", f"{self.dev.device}:COPY"): dest[:] = src def map(self, buf:Buffer): return buf.as_memoryview(force_zero_copy=True) def _offset(self, buf:memoryview, size:int, offset:int): return buf[offset:offset+size] diff --git a/tinygrad/runtime/support/hcq.py b/tinygrad/runtime/support/hcq.py index 62cce31367..f302eaf4d3 100644 --- a/tinygrad/runtime/support/hcq.py +++ b/tinygrad/runtime/support/hcq.py @@ -551,8 +551,8 @@ class HCQAllocatorBase(LRUAllocator[HCQDeviceType], Generic[HCQDeviceType]): self.b = copy_bufs or [self._alloc(batch_size, BufferSpec(host=True)) for _ in range(batch_cnt)] self.b_timeline, self.b_next, self.max_copyout_size = [0] * len(self.b), 0, max_copyout_size - def _map(self, buf:HCQBuffer): - if self.dev in buf.mapped_devs: return + def _map(self, buf:HCQBuffer) -> HCQBuffer: + if self.dev in buf.mapped_devs: return buf if buf.owner is None: raise RuntimeError(f"map failed: buffer {buf.va_addr} has no owner, it's a virtual buffer") if not hasattr(self, '_do_map'): raise NotImplementedError("map failed: no method implemented") @@ -560,6 +560,7 @@ class HCQAllocatorBase(LRUAllocator[HCQDeviceType], Generic[HCQDeviceType]): # Devices can save mappings and internal metadata as a new buffer. if (mb:=self._do_map(buf)) is not None: buf.mappings[self.dev] = mb buf.mapped_devs.append(self.dev) + return buf @suppress_finalizing def _free(self, buf:HCQBuffer, options:BufferSpec|None=None): diff --git a/tinygrad/tensor.py b/tinygrad/tensor.py index 0f10a0cda4..5da95e585a 100644 --- a/tinygrad/tensor.py +++ b/tinygrad/tensor.py @@ -218,7 +218,7 @@ class Tensor(RandMixin): # TODO: this is a hack for writing to DISK. remove with working assign if is_disk: - self._buffer().copyin(x._data()) + (b:=self._buffer()).copy_from(Buffer("PYTHON", b.size, b.dtype, opaque=x._data())) return self # STORE+AFTER: STORE is the write effect (void), AFTER wraps the view for correct shape/ranging assign = self.uop.after(self.uop.store(x.uop)) diff --git a/tinygrad/uop/ops.py b/tinygrad/uop/ops.py index 56458c0e25..56e3e511fc 100644 --- a/tinygrad/uop/ops.py +++ b/tinygrad/uop/ops.py @@ -770,8 +770,7 @@ class UOp(RandMixin, metaclass=UOpMetaClass): assert bdtype.fmt is not None, f"{bdtype=} has None fmt" ret = UOp.empty(shape:=get_shape(x), dtype=bdtype, device="PYTHON") data = struct.pack(f"{prod(shape)}{bdtype.fmt}", *[truncate[bdtype](bdtype.const(xi)) for xi in fully_flatten(x)]) - # fake realize. if target device is PYTHON it needs bytearray to be writable - ret.buffer.allocate(memoryview(data if device != "PYTHON" else bytearray(data))) + ret.buffer.allocate(memoryview(bytearray(data))) # fake realize. buffer storage must be writable, and bytes isn't if ret.dtype != dtype: ret = ret.cast(dtype) return ret if ret.device == device else ret.copy_to_device(device) def clone(self, device=None) -> UOp: