# bend-blas: cuBLAS on Array, one def per routine, plus device # buffers. Generated by gen/effs.py: edit the table there, not this file. # # The same conventions as blas.bend. cuBLAS is opened at runtime with # dlopen (libcudart, libcublas), so this file builds anywhere; on a # machine without CUDA every routine answers Fail{(1, ..)}, and # Cublas.available answers False. A device buffer is uploaded once # (Cublas.upload), referenced by its U32 id, never consumed, and freed # with Cublas.free; Cublas.sgemm_w reads one as its right operand, so a # model's weights cross the bus once. On the interpreter the JS twins # run on the host: plain loops, correct and slow. Not yet run on an # NVIDIA machine: the C is checked against the cuBLAS signatures only. import Base # True when cuBLAS opens on this machine: the answer of every other routine is Fail{(1, ..)} otherwise. Never fails. def Cublas.available() -> IO(Result<&1, &1, U32 & String, Bool>): import "./effs/cublas_available.c" import "./effs/cublas_available.js" def Cublas.available.try() -> IO(Bool): IO.try(Bool, Cublas.available()) # The first n floats of an array copied to the device; answers the buffer id. The array is consumed. def Cublas.upload(n: U32, a: Array) -> IO(Result<&1, &1, U32 & String, U32>): import "./effs/cublas_upload.c" import "./effs/cublas_upload.js" def Cublas.upload.try(n: U32, a: Array) -> IO(U32): IO.try(U32, Cublas.upload(n, a)) # The n floats of a device buffer back in a fresh array; the buffer stays. def Cublas.download(id: U32) -> IO(Result<&1, &1, U32 & String, Array>): import "./effs/cublas_download.c" import "./effs/cublas_download.js" def Cublas.download.try(id: U32) -> IO(Array): IO.try(Array, Cublas.download(id)) # Frees a device buffer; its id is dead after. def Cublas.free(id: U32) -> IO(Result<&1, &1, U32 & String, Unit>): import "./effs/cublas_free.c" import "./effs/cublas_free.js" def Cublas.free.try(id: U32) -> IO(Unit): IO.try(Unit, Cublas.free(id)) # C = alpha op(A) op(B) + beta C on the GPU, the operands copied to the # device for the call and the result back; the same arguments as # Blas.sgemm. def Cublas.sgemm(ta: U32, tb: U32, m: U32, n: U32, k: U32, alpha: F32, a: Array, b: Array, beta: F32, c: Array) -> IO(Result<&1, &1, U32 & String, Array>): import "./effs/cublas_sgemm.c" import "./effs/cublas_sgemm.js" def Cublas.sgemm.try(ta: U32, tb: U32, m: U32, n: U32, k: U32, alpha: F32, a: Array, b: Array, beta: F32, c: Array) -> IO(Array): IO.try(Array, Cublas.sgemm(ta, tb, m, n, k, alpha, a, b, beta, c)) # Cublas.sgemm with the right operand a device buffer (uploaded once with # Cublas.upload, never consumed): the shape of a model's weights. def Cublas.sgemm_w(ta: U32, tb: U32, m: U32, n: U32, k: U32, alpha: F32, a: Array, b: U32, beta: F32, c: Array) -> IO(Result<&1, &1, U32 & String, Array>): import "./effs/cublas_sgemm_w.c" import "./effs/cublas_sgemm_w.js" def Cublas.sgemm_w.try(ta: U32, tb: U32, m: U32, n: U32, k: U32, alpha: F32, a: Array, b: U32, beta: F32, c: Array) -> IO(Array): IO.try(Array, Cublas.sgemm_w(ta, tb, m, n, k, alpha, a, b, beta, c))