forked from tinygrad/tinygrad
* start hcq graph * hack-fix sync on amd * nv * fix nv * multigrah * fixes * temp fix for graph * this is not needed * fix * cleaner * linetr * fix none * faster cuda copy * faster amd copy * temp nv fixes * alloc on gpu * exp: faster amd * Revert "exp: faster amd" This reverts commit 2e4cfd1f7d8a33634c50fb5655cff1b40269d28c. * revert, unrelated * not in this pr * linter
133 lines
7.0 KiB
Python
133 lines
7.0 KiB
Python
from __future__ import annotations
|
|
from typing import Any, Optional, Dict, Tuple
|
|
import ctypes
|
|
from collections import defaultdict
|
|
from dataclasses import dataclass
|
|
from tinygrad.helpers import GlobalCounters, flat_mv, from_mv, getenv
|
|
from tinygrad.dtype import DType, ImageDType
|
|
|
|
@dataclass(frozen=True, eq=True)
|
|
class BufferOptions:
|
|
image: Optional[ImageDType] = None
|
|
uncached: bool = False
|
|
cpu_access: bool = False
|
|
host: bool = False
|
|
nolru: bool = False
|
|
|
|
class Buffer:
|
|
def __init__(self, device:str, size:int, dtype:DType, opaque:Any=None, options:Optional[BufferOptions]=None,
|
|
initial_value:Optional[bytes]=None, lb_refcount=0, base:Optional[Buffer]=None, offset:int=0, preallocate=False):
|
|
assert isinstance(dtype, DType)
|
|
if isinstance(dtype, ImageDType): options = BufferOptions(image=dtype) # TODO: image hack shouldn't be here. where should it be?
|
|
self.device, self.size, self.dtype, self.options, self.offset = device, size, dtype, options, offset
|
|
if base is None:
|
|
assert offset == 0, "base buffers can't have offset"
|
|
self._base = None
|
|
self._lb_refcount = lb_refcount
|
|
if opaque is not None: self.allocate(opaque)
|
|
if initial_value is not None:
|
|
self.allocate()
|
|
self.copyin(memoryview(initial_value))
|
|
else:
|
|
assert base._base is None, "base can't have a base"
|
|
assert device == base.device, "base must have the same device"
|
|
self._base = base
|
|
if preallocate: self.allocate()
|
|
@property
|
|
def base(self) -> Buffer: return self._base if self._base is not None else self
|
|
@property
|
|
def lb_refcount(self): return self.base._lb_refcount
|
|
def ref(self, cnt): self.base._lb_refcount += cnt
|
|
def is_allocated(self) -> bool: return hasattr(self, '_buf')
|
|
def ensure_allocated(self) -> Buffer: return self.allocate() if not hasattr(self, '_buf') else self
|
|
def allocate(self, opaque=None) -> Buffer:
|
|
assert not hasattr(self, '_buf'), "can't allocate already allocated buffer"
|
|
from tinygrad.device import Device
|
|
self.allocator = Device[self.device].allocator
|
|
if self._base is not None:
|
|
self._base.ensure_allocated()
|
|
assert hasattr(self.allocator, "offset"), "offset function required for view"
|
|
self._buf: Any = self.allocator.offset(self.base._buf, self.nbytes, self.offset)
|
|
else:
|
|
self._buf = opaque if opaque is not None else self.allocator.alloc(self.nbytes, self.options)
|
|
if not self.device.startswith("DISK"): GlobalCounters.mem_used += self.nbytes
|
|
return self
|
|
def __reduce__(self):
|
|
buf = None
|
|
if self._base is not None:
|
|
return self.__class__, (self.device, self.size, self.dtype, None, None, None, 0, self.base, self.offset, hasattr(self, '_buf'))
|
|
if self.device == "NPY": return self.__class__, (self.device, self.size, self.dtype, self._buf, self.options, None, self.lb_refcount)
|
|
if self.is_allocated():
|
|
buf = bytearray(self.nbytes)
|
|
self.copyout(memoryview(buf))
|
|
return self.__class__, (self.device, self.size, self.dtype, None, self.options, buf, self.lb_refcount)
|
|
@property
|
|
def nbytes(self): return self.size*self.dtype.itemsize
|
|
def __del__(self):
|
|
if not hasattr(self, '_buf'): return
|
|
if self._base is None:
|
|
if not self.device.startswith("DISK"): GlobalCounters.mem_used -= self.nbytes
|
|
self.allocator.free(self._buf, self.nbytes, self.options)
|
|
def __repr__(self):
|
|
return f"<buf real:{hasattr(self, '_buf')} device:{self.device} size:{self.size} dtype:{self.dtype}" + \
|
|
(f" offset:{self.offset}" if hasattr(self, "base") else "") + \
|
|
(">" if self.options is None else f" {self.options=}>")
|
|
def as_buffer(self, allow_zero_copy=False, force_zero_copy=False) -> memoryview:
|
|
# zero copy with as_buffer (disabled by default due to use after free)
|
|
if (force_zero_copy or allow_zero_copy) and hasattr(self.allocator, 'as_buffer'): return self.allocator.as_buffer(self._buf)
|
|
assert not force_zero_copy, "force zero copy was passed, but copy is required"
|
|
return self.copyout(memoryview(bytearray(self.nbytes)))
|
|
def copyin(self, mv:memoryview):
|
|
mv = flat_mv(mv)
|
|
assert len(mv) == self.nbytes, f"size mismatch, {len(mv)=} != {self.dtype=} {self.size=}"
|
|
assert self.is_allocated(), "can't copyin to unallocated buffer"
|
|
self.allocator.copyin(self._buf, mv)
|
|
return self
|
|
def copyout(self, mv:memoryview) -> memoryview:
|
|
mv = flat_mv(mv)
|
|
assert len(mv) == self.nbytes, f"size mismatch, {len(mv)=} != {self.dtype=} {self.size=}"
|
|
assert self.is_allocated(), "can't copyout unallocated buffer"
|
|
self.allocator.copyout(mv, self._buf)
|
|
return mv
|
|
def view(self, size:int, dtype:DType, offset:int) -> Buffer:
|
|
assert offset < self.nbytes, "offset must be less than nbytes"
|
|
if self._base is not None: return Buffer(self.device, size, dtype, base=self._base, offset=self.offset+offset)
|
|
return Buffer(self.device, size, dtype, base=self, offset=offset)
|
|
|
|
# TODO: size, dest, src are the same type. can we enforce this?
|
|
class Allocator:
|
|
def alloc(self, size:int, options:Optional[BufferOptions]=None):
|
|
assert not isinstance(size, int) or size > 0, f"alloc size must be positve, getting {size}"
|
|
return self._alloc(size, options if options is not None else BufferOptions())
|
|
def _alloc(self, size:int, options:BufferOptions): raise NotImplementedError("need alloc")
|
|
def free(self, opaque, size:int, options:Optional[BufferOptions]=None):
|
|
self._free(opaque, options if options is not None else BufferOptions())
|
|
def _free(self, opaque, options:BufferOptions): pass # if opaque is a Python object, you don't need a free
|
|
def copyin(self, dest, src:memoryview): raise NotImplementedError("need copyin")
|
|
def copyout(self, dest:memoryview, src): raise NotImplementedError("need copyout")
|
|
|
|
class LRUAllocator(Allocator): # pylint: disable=abstract-method
|
|
def __init__(self): self.cache: Dict[Tuple[int, Optional[BufferOptions]], Any] = defaultdict(list)
|
|
def alloc(self, size:int, options:Optional[BufferOptions]=None):
|
|
if len(c := self.cache[(size, options)]): return c.pop()
|
|
try: return super().alloc(size, options)
|
|
except (RuntimeError, MemoryError):
|
|
self.free_cache()
|
|
return super().alloc(size, options)
|
|
def free_cache(self):
|
|
for (sz,options),opaques in self.cache.items():
|
|
for opaque in opaques: super().free(opaque, sz, options)
|
|
opaques.clear()
|
|
def free(self, opaque:Any, size:int, options:Optional[BufferOptions]=None):
|
|
if getenv("LRU", 1) and (options is None or not options.nolru): self.cache[(size, options)].append(opaque)
|
|
else: super().free(opaque, size, options)
|
|
|
|
class _MallocAllocator(LRUAllocator):
|
|
def _alloc(self, size:int, options:BufferOptions): return (ctypes.c_uint8 * size)()
|
|
def as_buffer(self, src) -> memoryview: return flat_mv(memoryview(src))
|
|
def copyin(self, dest, src:memoryview): ctypes.memmove(dest, from_mv(src), len(src))
|
|
def copyout(self, dest:memoryview, src): ctypes.memmove(from_mv(dest), src, len(dest))
|
|
def offset(self, buf, size:int, offset:int): return from_mv(self.as_buffer(buf)[offset:offset+size])
|
|
|
|
MallocAllocator = _MallocAllocator()
|