diff --git a/src/ntops_lab/kernels/creation/affine_grid.py b/src/ntops_lab/kernels/creation/affine_grid.py index ec77e12..11c863e 100644 --- a/src/ntops_lab/kernels/creation/affine_grid.py +++ b/src/ntops_lab/kernels/creation/affine_grid.py @@ -12,24 +12,24 @@ def _component_kernel(height, width): scale_x = 2.0 / float(width) def arrangement(theta_row, out_component, scale_y_tensor, scale_x_tensor): - theta_arr = theta_row[:, None, None, :].expand((-1, height, width, -1)).flatten(end_dim=3) + theta_arr = theta_row[:, None, None, :, :].expand((-1, height, width, -1, -1)).flatten(end_dim=4) out_arr = out_component.flatten() return theta_arr.tile((1, 3)), out_arr.tile((1,)), scale_y_tensor, scale_x_tensor def application(theta, out, scale_y_tensor, scale_x_tensor): x = ((out.offsets(2) + 0.5) * scale_x_tensor) - 1.0 y = ((out.offsets(1) + 0.5) * scale_y_tensor) - 1.0 - t0 = ntl.sum(theta * (theta.offsets(1) == 0), axis=1) - t1 = ntl.sum(theta * (theta.offsets(1) == 1), axis=1) - t2 = ntl.sum(theta * (theta.offsets(1) == 2), axis=1) + t0 = ntl.sum(theta * (theta.offsets(2) == 0), axis=1) + t1 = ntl.sum(theta * (theta.offsets(2) == 1), axis=1) + t2 = ntl.sum(theta * (theta.offsets(2) == 2), axis=1) out = t0 * x + t1 * y + t2 return ninetoothed.make( arrangement, application, ( - Tensor(2), Tensor(3), + Tensor(4), Tensor(0, constexpr=True, value=scale_y, name="scale_y"), Tensor(0, constexpr=True, value=scale_x, name="scale_x"), ), @@ -46,6 +46,5 @@ def run(theta, size, align_corners=False): kernel = _component_kernel(height, width) scale_y = 2.0 / float(height) scale_x = 2.0 / float(width) - kernel(theta[:, 0, :], out[..., 0], scale_y, scale_x) - kernel(theta[:, 1, :], out[..., 1], scale_y, scale_x) + kernel(theta, out, scale_y, scale_x) return out diff --git a/src/ntops_lab/kernels/fused/general/_flash_attention.py b/src/ntops_lab/kernels/fused/general/_flash_attention.py index b793afd..84d64e9 100644 --- a/src/ntops_lab/kernels/fused/general/_flash_attention.py +++ b/src/ntops_lab/kernels/fused/general/_flash_attention.py @@ -35,7 +35,7 @@ def application(q, k, v, is_causal, softmax_scale, out): m_i = ntl.full((q.shape[-2],), float("-inf"), dtype=ntl.float32) for i in range(k.shape[0]): - qk = ntl.dot(q_loaded, ntl.trans(k[i])) + qk = ntl.dot(q_loaded, ntl.trans(k[i])).to(ntl.float32) qk = ntl.where(k[i].offsets(-2) < k.source.shape[-2], qk, float("-inf")) if is_causal: diff --git a/src/ntops_lab/kernels/fused/general/add_layer_norm_sigmoid.py b/src/ntops_lab/kernels/fused/general/add_layer_norm_sigmoid.py index 2107a39..aed3260 100644 --- a/src/ntops_lab/kernels/fused/general/add_layer_norm_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/add_layer_norm_sigmoid.py @@ -16,7 +16,7 @@ def application(x, residual, gamma, beta, out, hidden): centered = y - mean[:, None] var = ntl.sum(centered * centered, axis=1) / hidden value = centered * ntl.rsqrt(var[:, None] + 1.0e-5) * gamma + beta - out = ntl.sigmoid(value) + out = (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/add_layer_norm_silu.py b/src/ntops_lab/kernels/fused/general/add_layer_norm_silu.py index 35e3c54..6d4d725 100644 --- a/src/ntops_lab/kernels/fused/general/add_layer_norm_silu.py +++ b/src/ntops_lab/kernels/fused/general/add_layer_norm_silu.py @@ -16,7 +16,7 @@ def application(x, residual, gamma, beta, out, hidden): centered = y - mean[:, None] var = ntl.sum(centered * centered, axis=1) / hidden value = centered * ntl.rsqrt(var[:, None] + 1.0e-5) * gamma + beta - out = value * ntl.sigmoid(value) + out = value * (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/add_rms_norm_sigmoid.py b/src/ntops_lab/kernels/fused/general/add_rms_norm_sigmoid.py index 2a74def..8e2e023 100644 --- a/src/ntops_lab/kernels/fused/general/add_rms_norm_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/add_rms_norm_sigmoid.py @@ -13,7 +13,7 @@ def application(x, residual, weight, out, hidden): y = x + residual rrms = ntl.rsqrt(ntl.sum(y * y, axis=1) / hidden + 1.0e-5) value = y * rrms[:, None] * weight - out = ntl.sigmoid(value) + out = (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/add_rms_norm_silu.py b/src/ntops_lab/kernels/fused/general/add_rms_norm_silu.py index 9752ff9..4a8ed80 100644 --- a/src/ntops_lab/kernels/fused/general/add_rms_norm_silu.py +++ b/src/ntops_lab/kernels/fused/general/add_rms_norm_silu.py @@ -13,7 +13,7 @@ def application(x, residual, weight, out, hidden): y = x + residual rrms = ntl.rsqrt(ntl.sum(y * y, axis=1) / hidden + 1.0e-5) value = y * rrms[:, None] * weight - out = value * ntl.sigmoid(value) + out = value * (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_abs.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_abs.py index 751d44f..eeb105c 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_abs.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_abs.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_clamp.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_clamp.py index d608412..044b65a 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_clamp.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_clamp.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_cube.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_cube.py index f57abe2..91a4466 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_cube.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_cube.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_elu.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_elu.py index 1099136..0285a47 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_elu.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_elu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_gelu.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_gelu.py index 2fe4712..7f75b15 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_gelu.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_gelu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardshrink.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardshrink.py index a1add31..4b21c3e 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardshrink.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardsigmoid.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardsigmoid.py index 8ef48eb..040cb71 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardsigmoid.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardswish.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardswish.py index 9cc9bbf..dff59e6 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardswish.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardswish.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardtanh.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardtanh.py index c059dda..e70b36e 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardtanh.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_hardtanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_leaky_relu.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_leaky_relu.py index e4b9a68..80bad71 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_leaky_relu.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_leaky_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_logsigmoid.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_logsigmoid.py index 203bb4a..5db1759 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_logsigmoid.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_logsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_neg.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_neg.py index 17d8124..1e118be 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_neg.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_neg.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu.py index b2ed592..2984b74 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu6.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu6.py index 2a9d618..c41dfa7 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu6.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_relu6.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_sigmoid.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_sigmoid.py index 927061f..6c5acff 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_sigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): @@ -22,7 +22,7 @@ def application(c, a, b, bias, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = c + acc + bias - out = (ntl.sigmoid(value)).to(ntl.float16) + out = ((1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_fused_addmm_bias_sigmoid") diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_silu.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_silu.py index 0866005..2aac43f 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_silu.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_silu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): @@ -22,7 +22,7 @@ def application(c, a, b, bias, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = c + acc + bias - out = (value * ntl.sigmoid(value)).to(ntl.float16) + out = (value * (1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_fused_addmm_bias_silu") diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softplus.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softplus.py index 360acea..2cba983 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softplus.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softplus.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softshrink.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softshrink.py index 4c9492c..0fdb702 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softshrink.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softsign.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softsign.py index d1f848f..a51b7fb 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softsign.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_softsign.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_square.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_square.py index 79c4cab..7158759 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_square.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_square.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanh.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanh.py index c0d8dfd..d57fd14 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanh.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanhshrink.py b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanhshrink.py index 3a39424..3b9d409 100644 --- a/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanhshrink.py +++ b/src/ntops_lab/kernels/fused/general/fused_addmm_bias_tanhshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -14,7 +14,7 @@ def arrangement(c, a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return c_arr, a_arr, b_arr, bias_arr, out_arr def application(c, a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_abs.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_abs.py index 9b599c8..26f604c 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_abs.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_abs.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_clamp.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_clamp.py index 41e2507..db7e3d0 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_clamp.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_clamp.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_cube.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_cube.py index adcad72..d4c652e 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_cube.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_cube.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_elu.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_elu.py index a60f437..79b1d34 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_elu.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_elu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_gelu.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_gelu.py index 9a392a8..09e97f6 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_gelu.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_gelu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardshrink.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardshrink.py index 3f0fb5f..18a0952 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardshrink.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardsigmoid.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardsigmoid.py index 0fd091d..e10b7ef 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardsigmoid.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardswish.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardswish.py index 30aba52..ca97db7 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardswish.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardswish.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardtanh.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardtanh.py index 849f01f..6ebe132 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardtanh.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_hardtanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_leaky_relu.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_leaky_relu.py index 28a51d8..15fe207 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_leaky_relu.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_leaky_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_logsigmoid.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_logsigmoid.py index 9c611ef..e8efd09 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_logsigmoid.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_logsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_neg.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_neg.py index eebcf68..b2cacaa 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_neg.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_neg.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu.py index 29dc143..2d9a8c5 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu6.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu6.py index 751ab49..f835ace 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu6.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_relu6.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_sigmoid.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_sigmoid.py index a304a90..bffd515 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_sigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): @@ -21,7 +21,7 @@ def application(a, b, bias, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = acc + bias - out = (ntl.sigmoid(value)).to(ntl.float16) + out = ((1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_fused_gemm_bias_sigmoid") diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_silu.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_silu.py index f220937..ab5951a 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_silu.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_silu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): @@ -21,7 +21,7 @@ def application(a, b, bias, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = acc + bias - out = (value * ntl.sigmoid(value)).to(ntl.float16) + out = (value * (1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_fused_gemm_bias_silu") diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softplus.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softplus.py index dc8c205..127b856 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softplus.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softplus.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softshrink.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softshrink.py index 3f21b10..cd40b83 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softshrink.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softsign.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softsign.py index b2aebc3..29ab697 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softsign.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_softsign.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_square.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_square.py index e9b9474..029ff02 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_square.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_square.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanh.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanh.py index cf8376c..629b390 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanh.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanhshrink.py b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanhshrink.py index deff152..b8900d4 100644 --- a/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanhshrink.py +++ b/src/ntops_lab/kernels/fused/general/fused_gemm_bias_tanhshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/fused/general/gate_sigmoid_add_mul.py b/src/ntops_lab/kernels/fused/general/gate_sigmoid_add_mul.py index 96095d4..f2287ba 100644 --- a/src/ntops_lab/kernels/fused/general/gate_sigmoid_add_mul.py +++ b/src/ntops_lab/kernels/fused/general/gate_sigmoid_add_mul.py @@ -9,7 +9,7 @@ def arrangement(a, b, c, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), c.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, c, out): - out = ntl.sigmoid((a + b)) * c + out = (1.0 / (1.0 + ntl.exp(-(a + b)))) * c kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_gate_sigmoid_add_mul") diff --git a/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul.py b/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul.py index 81ef176..9293d3b 100644 --- a/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul.py +++ b/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul.py @@ -9,7 +9,7 @@ def arrangement(a, b, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, out): - out = ntl.sigmoid(a) * b + out = (1.0 / (1.0 + ntl.exp(-(a)))) * b kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_gate_sigmoid_mul") diff --git a/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul_add.py b/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul_add.py index e2ae3be..c2dc53d 100644 --- a/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul_add.py +++ b/src/ntops_lab/kernels/fused/general/gate_sigmoid_mul_add.py @@ -9,7 +9,7 @@ def arrangement(a, b, c, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), c.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, c, out): - out = ntl.sigmoid(a) * b + c + out = (1.0 / (1.0 + ntl.exp(-(a)))) * b + c kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_gate_sigmoid_mul_add") diff --git a/src/ntops_lab/kernels/fused/general/gate_silu_add_mul.py b/src/ntops_lab/kernels/fused/general/gate_silu_add_mul.py index 4b99d36..dbed3ad 100644 --- a/src/ntops_lab/kernels/fused/general/gate_silu_add_mul.py +++ b/src/ntops_lab/kernels/fused/general/gate_silu_add_mul.py @@ -9,7 +9,7 @@ def arrangement(a, b, c, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), c.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, c, out): - out = (a + b) * ntl.sigmoid((a + b)) * c + out = (a + b) * (1.0 / (1.0 + ntl.exp(-(a + b)))) * c kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_gate_silu_add_mul") diff --git a/src/ntops_lab/kernels/fused/general/gate_silu_mul.py b/src/ntops_lab/kernels/fused/general/gate_silu_mul.py index 8785e16..e34d445 100644 --- a/src/ntops_lab/kernels/fused/general/gate_silu_mul.py +++ b/src/ntops_lab/kernels/fused/general/gate_silu_mul.py @@ -9,7 +9,7 @@ def arrangement(a, b, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, out): - out = a * ntl.sigmoid(a) * b + out = a * (1.0 / (1.0 + ntl.exp(-(a)))) * b kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_gate_silu_mul") diff --git a/src/ntops_lab/kernels/fused/general/gate_silu_mul_add.py b/src/ntops_lab/kernels/fused/general/gate_silu_mul_add.py index cd7d42a..6bf551d 100644 --- a/src/ntops_lab/kernels/fused/general/gate_silu_mul_add.py +++ b/src/ntops_lab/kernels/fused/general/gate_silu_mul_add.py @@ -9,7 +9,7 @@ def arrangement(a, b, c, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), c.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, c, out): - out = a * ntl.sigmoid(a) * b + c + out = a * (1.0 / (1.0 + ntl.exp(-(a)))) * b + c kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_gate_silu_mul_add") diff --git a/src/ntops_lab/kernels/fused/general/glu.py b/src/ntops_lab/kernels/fused/general/glu.py index 278a775..876c44a 100644 --- a/src/ntops_lab/kernels/fused/general/glu.py +++ b/src/ntops_lab/kernels/fused/general/glu.py @@ -9,7 +9,7 @@ def arrangement(a, b, out): return a.flatten().tile((BLOCK,)), b.flatten().tile((BLOCK,)), out.flatten().tile((BLOCK,)) def application(a, b, out): - out = a * ntl.sigmoid(b) + out = a * (1.0 / (1.0 + ntl.exp(-(b)))) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_glu") diff --git a/src/ntops_lab/kernels/fused/general/layer_norm_sigmoid.py b/src/ntops_lab/kernels/fused/general/layer_norm_sigmoid.py index 654fba0..def393d 100644 --- a/src/ntops_lab/kernels/fused/general/layer_norm_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/layer_norm_sigmoid.py @@ -16,7 +16,7 @@ def application(x, gamma, beta, out, hidden): centered = y - mean[:, None] var = ntl.sum(centered * centered, axis=1) / hidden value = centered * ntl.rsqrt(var[:, None] + 1.0e-5) * gamma + beta - out = ntl.sigmoid(value) + out = (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/layer_norm_silu.py b/src/ntops_lab/kernels/fused/general/layer_norm_silu.py index 8aa9db8..84cdcc4 100644 --- a/src/ntops_lab/kernels/fused/general/layer_norm_silu.py +++ b/src/ntops_lab/kernels/fused/general/layer_norm_silu.py @@ -16,7 +16,7 @@ def application(x, gamma, beta, out, hidden): centered = y - mean[:, None] var = ntl.sum(centered * centered, axis=1) / hidden value = centered * ntl.rsqrt(var[:, None] + 1.0e-5) * gamma + beta - out = value * ntl.sigmoid(value) + out = value * (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/residual_bias_sigmoid.py b/src/ntops_lab/kernels/fused/general/residual_bias_sigmoid.py index 9bcedf3..97ca3ec 100644 --- a/src/ntops_lab/kernels/fused/general/residual_bias_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/residual_bias_sigmoid.py @@ -9,7 +9,7 @@ def arrangement(a, b, c, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), c.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, c, out): - out = ntl.sigmoid((a + b + c)) + out = (1.0 / (1.0 + ntl.exp(-(a + b + c)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_residual_bias_sigmoid") diff --git a/src/ntops_lab/kernels/fused/general/residual_bias_silu.py b/src/ntops_lab/kernels/fused/general/residual_bias_silu.py index d24a461..d210983 100644 --- a/src/ntops_lab/kernels/fused/general/residual_bias_silu.py +++ b/src/ntops_lab/kernels/fused/general/residual_bias_silu.py @@ -9,7 +9,7 @@ def arrangement(a, b, c, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), c.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, c, out): - out = (a + b + c) * ntl.sigmoid((a + b + c)) + out = (a + b + c) * (1.0 / (1.0 + ntl.exp(-(a + b + c)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_residual_bias_silu") diff --git a/src/ntops_lab/kernels/fused/general/residual_sigmoid.py b/src/ntops_lab/kernels/fused/general/residual_sigmoid.py index 1963583..bc3dfa8 100644 --- a/src/ntops_lab/kernels/fused/general/residual_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/residual_sigmoid.py @@ -9,7 +9,7 @@ def arrangement(a, b, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, out): - out = ntl.sigmoid((a + b)) + out = (1.0 / (1.0 + ntl.exp(-(a + b)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_residual_sigmoid") diff --git a/src/ntops_lab/kernels/fused/general/residual_silu.py b/src/ntops_lab/kernels/fused/general/residual_silu.py index 3d72203..55ba8ba 100644 --- a/src/ntops_lab/kernels/fused/general/residual_silu.py +++ b/src/ntops_lab/kernels/fused/general/residual_silu.py @@ -9,7 +9,7 @@ def arrangement(a, b, out): return a.tile((BLOCK_SIZE,)), b.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(a, b, out): - out = (a + b) * ntl.sigmoid((a + b)) + out = (a + b) * (1.0 / (1.0 + ntl.exp(-(a + b)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_residual_silu") diff --git a/src/ntops_lab/kernels/fused/general/rms_norm_sigmoid.py b/src/ntops_lab/kernels/fused/general/rms_norm_sigmoid.py index afc8db0..c35f277 100644 --- a/src/ntops_lab/kernels/fused/general/rms_norm_sigmoid.py +++ b/src/ntops_lab/kernels/fused/general/rms_norm_sigmoid.py @@ -13,7 +13,7 @@ def application(x, weight, out, hidden): y = x rrms = ntl.rsqrt(ntl.sum(y * y, axis=1) / hidden + 1.0e-5) value = y * rrms[:, None] * weight - out = ntl.sigmoid(value) + out = (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/fused/general/rms_norm_silu.py b/src/ntops_lab/kernels/fused/general/rms_norm_silu.py index 069845c..476a93a 100644 --- a/src/ntops_lab/kernels/fused/general/rms_norm_silu.py +++ b/src/ntops_lab/kernels/fused/general/rms_norm_silu.py @@ -13,7 +13,7 @@ def application(x, weight, out, hidden): y = x rrms = ntl.rsqrt(ntl.sum(y * y, axis=1) / hidden + 1.0e-5) value = y * rrms[:, None] * weight - out = value * ntl.sigmoid(value) + out = value * (1.0 / (1.0 + ntl.exp(-(value)))) @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/layout/conv1d.py b/src/ntops_lab/kernels/layout/conv1d.py index 01e4d4c..5494451 100644 --- a/src/ntops_lab/kernels/layout/conv1d.py +++ b/src/ntops_lab/kernels/layout/conv1d.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/conv2d.py b/src/ntops_lab/kernels/layout/conv2d.py index 15b0e87..0e2e0fa 100644 --- a/src/ntops_lab/kernels/layout/conv2d.py +++ b/src/ntops_lab/kernels/layout/conv2d.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/conv3d.py b/src/ntops_lab/kernels/layout/conv3d.py index 80875ba..56a8237 100644 --- a/src/ntops_lab/kernels/layout/conv3d.py +++ b/src/ntops_lab/kernels/layout/conv3d.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/conv_depthwise2d.py b/src/ntops_lab/kernels/layout/conv_depthwise2d.py index 0acf3bc..96a80ee 100644 --- a/src/ntops_lab/kernels/layout/conv_depthwise2d.py +++ b/src/ntops_lab/kernels/layout/conv_depthwise2d.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/conv_transpose1d.py b/src/ntops_lab/kernels/layout/conv_transpose1d.py index dd989ec..08aa7e9 100644 --- a/src/ntops_lab/kernels/layout/conv_transpose1d.py +++ b/src/ntops_lab/kernels/layout/conv_transpose1d.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/conv_transpose2d.py b/src/ntops_lab/kernels/layout/conv_transpose2d.py index 4aa6f51..af0231f 100644 --- a/src/ntops_lab/kernels/layout/conv_transpose2d.py +++ b/src/ntops_lab/kernels/layout/conv_transpose2d.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/cudnn_convolution.py b/src/ntops_lab/kernels/layout/cudnn_convolution.py index 723595c..b90beb5 100644 --- a/src/ntops_lab/kernels/layout/cudnn_convolution.py +++ b/src/ntops_lab/kernels/layout/cudnn_convolution.py @@ -8,7 +8,7 @@ INPUT_PRECISION_IEEE = 2 -def _mm_arrangement(input, other, output, input_precision, block_size_m=16, block_size_n=16, block_size_k=16): +def _mm_arrangement(input, other, output, input_precision, block_size_m=32, block_size_n=32, block_size_k=16): output_arranged = output.tile((block_size_m, block_size_n)) input_arranged = input.tile((block_size_m, block_size_k)) @@ -48,7 +48,7 @@ def _conv2d_arrangement(input, weight, bias, output, input_precision, pad_h, pad output_arranged = output.permute((0, 2, 3, 1)).flatten(end_dim=3) - bias_arranged = bias_arranged.tile((16, 16)) + bias_arranged = bias_arranged.tile((32, 32)) input_arranged, weight_arranged, output_arranged, input_precision_arranged = _mm_arrangement(input_arranged, weight_arranged, output_arranged, input_precision) return input_arranged, weight_arranged, bias_arranged, output_arranged, input_precision_arranged diff --git a/src/ntops_lab/kernels/layout/upsample_linear1d.py b/src/ntops_lab/kernels/layout/upsample_linear1d.py index 7c17f92..38a61d5 100644 --- a/src/ntops_lab/kernels/layout/upsample_linear1d.py +++ b/src/ntops_lab/kernels/layout/upsample_linear1d.py @@ -25,16 +25,16 @@ def arrangement(x, out): def application(x, out): for i in range(16): - prev = x[i] - cur = x[i + 1] - nxt = x[i + 2] + prev = x[i].to(ntl.float32) + cur = x[i + 1].to(ntl.float32) + nxt = x[i + 2].to(ntl.float32) valid = x[i + 1].offsets(-1) < x.source.shape[-1] first = x[i + 1].offsets(-1) == 0 last = x[i + 1].offsets(-1) == x.source.shape[-1] - 1 even = ntl.where(first, cur, prev * 0.25 + cur * 0.75) odd = ntl.where(last, cur, cur * 0.75 + nxt * 0.25) - out[2 * i] = ntl.where(valid, even, out[2 * i]) - out[2 * i + 1] = ntl.where(valid, odd, out[2 * i + 1]) + out[2 * i] = ntl.where(valid, even, out[2 * i].to(ntl.float32)) + out[2 * i + 1] = ntl.where(valid, odd, out[2 * i + 1].to(ntl.float32)) kernel = ninetoothed.make(arrangement, application, (Tensor(3, other=0.0), Tensor(3)), kernel_name="ntops_lab_upsample_linear1d_scale2_align_false", max_num_configs=1) diff --git a/src/ntops_lab/kernels/linear/addbmm.py b/src/ntops_lab/kernels/linear/addbmm.py index a7d3475..91ce288 100644 --- a/src/ntops_lab/kernels/linear/addbmm.py +++ b/src/ntops_lab/kernels/linear/addbmm.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm.py b/src/ntops_lab/kernels/linear/addmm.py index daab917..3473c5d 100644 --- a/src/ntops_lab/kernels/linear/addmm.py +++ b/src/ntops_lab/kernels/linear/addmm.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_abs.py b/src/ntops_lab/kernels/linear/addmm_abs.py index 913b449..50435ee 100644 --- a/src/ntops_lab/kernels/linear/addmm_abs.py +++ b/src/ntops_lab/kernels/linear/addmm_abs.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_cube.py b/src/ntops_lab/kernels/linear/addmm_cube.py index 1b3ab89..2d3ffe5 100644 --- a/src/ntops_lab/kernels/linear/addmm_cube.py +++ b/src/ntops_lab/kernels/linear/addmm_cube.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_elu.py b/src/ntops_lab/kernels/linear/addmm_elu.py index e18053b..e4d504b 100644 --- a/src/ntops_lab/kernels/linear/addmm_elu.py +++ b/src/ntops_lab/kernels/linear/addmm_elu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_gelu.py b/src/ntops_lab/kernels/linear/addmm_gelu.py index d9305bb..1b3e2b1 100644 --- a/src/ntops_lab/kernels/linear/addmm_gelu.py +++ b/src/ntops_lab/kernels/linear/addmm_gelu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_hardshrink.py b/src/ntops_lab/kernels/linear/addmm_hardshrink.py index 504b823..4cc7cae 100644 --- a/src/ntops_lab/kernels/linear/addmm_hardshrink.py +++ b/src/ntops_lab/kernels/linear/addmm_hardshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_hardsigmoid.py b/src/ntops_lab/kernels/linear/addmm_hardsigmoid.py index 27e9ee6..8add677 100644 --- a/src/ntops_lab/kernels/linear/addmm_hardsigmoid.py +++ b/src/ntops_lab/kernels/linear/addmm_hardsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_hardtanh.py b/src/ntops_lab/kernels/linear/addmm_hardtanh.py index 3a799ff..752aed7 100644 --- a/src/ntops_lab/kernels/linear/addmm_hardtanh.py +++ b/src/ntops_lab/kernels/linear/addmm_hardtanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_leaky_relu.py b/src/ntops_lab/kernels/linear/addmm_leaky_relu.py index 04752af..d096c3d 100644 --- a/src/ntops_lab/kernels/linear/addmm_leaky_relu.py +++ b/src/ntops_lab/kernels/linear/addmm_leaky_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_logsigmoid.py b/src/ntops_lab/kernels/linear/addmm_logsigmoid.py index 2487446..f70dc86 100644 --- a/src/ntops_lab/kernels/linear/addmm_logsigmoid.py +++ b/src/ntops_lab/kernels/linear/addmm_logsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_neg.py b/src/ntops_lab/kernels/linear/addmm_neg.py index 638efad..c5ff181 100644 --- a/src/ntops_lab/kernels/linear/addmm_neg.py +++ b/src/ntops_lab/kernels/linear/addmm_neg.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_relu.py b/src/ntops_lab/kernels/linear/addmm_relu.py index 6257ad8..5744044 100644 --- a/src/ntops_lab/kernels/linear/addmm_relu.py +++ b/src/ntops_lab/kernels/linear/addmm_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_relu6.py b/src/ntops_lab/kernels/linear/addmm_relu6.py index 3951bfb..7c53b2f 100644 --- a/src/ntops_lab/kernels/linear/addmm_relu6.py +++ b/src/ntops_lab/kernels/linear/addmm_relu6.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_sigmoid.py b/src/ntops_lab/kernels/linear/addmm_sigmoid.py index da3df3b..955107e 100644 --- a/src/ntops_lab/kernels/linear/addmm_sigmoid.py +++ b/src/ntops_lab/kernels/linear/addmm_sigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) @@ -21,7 +21,7 @@ def application(c, a, b, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = c + acc - out = (ntl.sigmoid(value)).to(ntl.float16) + out = ((1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_addmm_sigmoid") diff --git a/src/ntops_lab/kernels/linear/addmm_silu.py b/src/ntops_lab/kernels/linear/addmm_silu.py index e578939..3c14224 100644 --- a/src/ntops_lab/kernels/linear/addmm_silu.py +++ b/src/ntops_lab/kernels/linear/addmm_silu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) @@ -21,7 +21,7 @@ def application(c, a, b, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = c + acc - out = (value * ntl.sigmoid(value)).to(ntl.float16) + out = (value * (1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_addmm_silu") diff --git a/src/ntops_lab/kernels/linear/addmm_softshrink.py b/src/ntops_lab/kernels/linear/addmm_softshrink.py index f569099..d66e1eb 100644 --- a/src/ntops_lab/kernels/linear/addmm_softshrink.py +++ b/src/ntops_lab/kernels/linear/addmm_softshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_square.py b/src/ntops_lab/kernels/linear/addmm_square.py index 11875a1..535c889 100644 --- a/src/ntops_lab/kernels/linear/addmm_square.py +++ b/src/ntops_lab/kernels/linear/addmm_square.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_tanh.py b/src/ntops_lab/kernels/linear/addmm_tanh.py index 7fbbbc5..d3c1281 100644 --- a/src/ntops_lab/kernels/linear/addmm_tanh.py +++ b/src/ntops_lab/kernels/linear/addmm_tanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmm_tanhshrink.py b/src/ntops_lab/kernels/linear/addmm_tanhshrink.py index 90d5a77..431eec1 100644 --- a/src/ntops_lab/kernels/linear/addmm_tanhshrink.py +++ b/src/ntops_lab/kernels/linear/addmm_tanhshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/addmv.py b/src/ntops_lab/kernels/linear/addmv.py index 19ebf42..68bf498 100644 --- a/src/ntops_lab/kernels/linear/addmv.py +++ b/src/ntops_lab/kernels/linear/addmv.py @@ -1,15 +1,14 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor -BM = block_size() -BK = block_size() +BM = 1 +BK = -1 def arrangement(bias, a, x, out): a_arr = a.tile((BM, BK)) - x_arr = x.tile((BK,)) - x_arr = x_arr.expand((a_arr.shape[0], -1)) + x_arr = x[None, :].expand((a.shape[0], -1)).tile((BM, BK)) out_arr = out.tile((BM,)) bias_arr = bias.tile((BM,)) return bias_arr, a_arr, x_arr, out_arr diff --git a/src/ntops_lab/kernels/linear/baddbmm.py b/src/ntops_lab/kernels/linear/baddbmm.py index 0ea44d1..816c231 100644 --- a/src/ntops_lab/kernels/linear/baddbmm.py +++ b/src/ntops_lab/kernels/linear/baddbmm.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(c, a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm.py b/src/ntops_lab/kernels/linear/bmm.py index a7cf9bf..36a50ca 100644 --- a/src/ntops_lab/kernels/linear/bmm.py +++ b/src/ntops_lab/kernels/linear/bmm.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_abs.py b/src/ntops_lab/kernels/linear/bmm_abs.py index 9b2519d..34c9432 100644 --- a/src/ntops_lab/kernels/linear/bmm_abs.py +++ b/src/ntops_lab/kernels/linear/bmm_abs.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_cube.py b/src/ntops_lab/kernels/linear/bmm_cube.py index 44fe316..a1e0869 100644 --- a/src/ntops_lab/kernels/linear/bmm_cube.py +++ b/src/ntops_lab/kernels/linear/bmm_cube.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_elu.py b/src/ntops_lab/kernels/linear/bmm_elu.py index 94f8e85..b447610 100644 --- a/src/ntops_lab/kernels/linear/bmm_elu.py +++ b/src/ntops_lab/kernels/linear/bmm_elu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_gelu.py b/src/ntops_lab/kernels/linear/bmm_gelu.py index d58bf80..c3e5a52 100644 --- a/src/ntops_lab/kernels/linear/bmm_gelu.py +++ b/src/ntops_lab/kernels/linear/bmm_gelu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_hardshrink.py b/src/ntops_lab/kernels/linear/bmm_hardshrink.py index 41fccd3..669ad34 100644 --- a/src/ntops_lab/kernels/linear/bmm_hardshrink.py +++ b/src/ntops_lab/kernels/linear/bmm_hardshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) @@ -19,7 +19,7 @@ def application(a, b, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - value = acc.to(ntl.float16) + value = acc.to(ntl.float16).to(ntl.float32) out = (ntl.where(value > 0.5, value, ntl.where(value < -0.5, value, 0.0))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_bmm_hardshrink") diff --git a/src/ntops_lab/kernels/linear/bmm_hardsigmoid.py b/src/ntops_lab/kernels/linear/bmm_hardsigmoid.py index 3a94917..963b358 100644 --- a/src/ntops_lab/kernels/linear/bmm_hardsigmoid.py +++ b/src/ntops_lab/kernels/linear/bmm_hardsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_hardtanh.py b/src/ntops_lab/kernels/linear/bmm_hardtanh.py index d1c35c9..862649a 100644 --- a/src/ntops_lab/kernels/linear/bmm_hardtanh.py +++ b/src/ntops_lab/kernels/linear/bmm_hardtanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_leaky_relu.py b/src/ntops_lab/kernels/linear/bmm_leaky_relu.py index b97ce90..34354b2 100644 --- a/src/ntops_lab/kernels/linear/bmm_leaky_relu.py +++ b/src/ntops_lab/kernels/linear/bmm_leaky_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_logsigmoid.py b/src/ntops_lab/kernels/linear/bmm_logsigmoid.py index 4d0e05b..e8349cc 100644 --- a/src/ntops_lab/kernels/linear/bmm_logsigmoid.py +++ b/src/ntops_lab/kernels/linear/bmm_logsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_neg.py b/src/ntops_lab/kernels/linear/bmm_neg.py index 3a19845..661cce1 100644 --- a/src/ntops_lab/kernels/linear/bmm_neg.py +++ b/src/ntops_lab/kernels/linear/bmm_neg.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_out.py b/src/ntops_lab/kernels/linear/bmm_out.py index 9a25ab8..595e408 100644 --- a/src/ntops_lab/kernels/linear/bmm_out.py +++ b/src/ntops_lab/kernels/linear/bmm_out.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_relu.py b/src/ntops_lab/kernels/linear/bmm_relu.py index 5e651b6..a1d7cc6 100644 --- a/src/ntops_lab/kernels/linear/bmm_relu.py +++ b/src/ntops_lab/kernels/linear/bmm_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_relu6.py b/src/ntops_lab/kernels/linear/bmm_relu6.py index 0ae7be6..47641d9 100644 --- a/src/ntops_lab/kernels/linear/bmm_relu6.py +++ b/src/ntops_lab/kernels/linear/bmm_relu6.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_sigmoid.py b/src/ntops_lab/kernels/linear/bmm_sigmoid.py index fc30319..6694388 100644 --- a/src/ntops_lab/kernels/linear/bmm_sigmoid.py +++ b/src/ntops_lab/kernels/linear/bmm_sigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) @@ -23,7 +23,7 @@ def application(a, b, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - out = (ntl.sigmoid(acc)).to(ntl.float16) + out = ((1.0 / (1.0 + ntl.exp(-(acc))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_bmm_sigmoid_2d") diff --git a/src/ntops_lab/kernels/linear/bmm_silu.py b/src/ntops_lab/kernels/linear/bmm_silu.py index 75882a7..085b90d 100644 --- a/src/ntops_lab/kernels/linear/bmm_silu.py +++ b/src/ntops_lab/kernels/linear/bmm_silu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) @@ -23,7 +23,7 @@ def application(a, b, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - out = (acc * ntl.sigmoid(acc)).to(ntl.float16) + out = (acc * (1.0 / (1.0 + ntl.exp(-(acc))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_bmm_silu_2d") diff --git a/src/ntops_lab/kernels/linear/bmm_softshrink.py b/src/ntops_lab/kernels/linear/bmm_softshrink.py index 689867c..ef48494 100644 --- a/src/ntops_lab/kernels/linear/bmm_softshrink.py +++ b/src/ntops_lab/kernels/linear/bmm_softshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_square.py b/src/ntops_lab/kernels/linear/bmm_square.py index bad774e..3255388 100644 --- a/src/ntops_lab/kernels/linear/bmm_square.py +++ b/src/ntops_lab/kernels/linear/bmm_square.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_tanh.py b/src/ntops_lab/kernels/linear/bmm_tanh.py index a393dda..70e7035 100644 --- a/src/ntops_lab/kernels/linear/bmm_tanh.py +++ b/src/ntops_lab/kernels/linear/bmm_tanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/bmm_tanhshrink.py b/src/ntops_lab/kernels/linear/bmm_tanhshrink.py index 185a175..f40a986 100644 --- a/src/ntops_lab/kernels/linear/bmm_tanhshrink.py +++ b/src/ntops_lab/kernels/linear/bmm_tanhshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/dot.py b/src/ntops_lab/kernels/linear/dot.py index b3d9f4d..446ff09 100644 --- a/src/ntops_lab/kernels/linear/dot.py +++ b/src/ntops_lab/kernels/linear/dot.py @@ -1,15 +1,14 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor -BLOCK_N = block_size() def arrangement(x, y, out): - return x.tile((BLOCK_N,)), y.tile((BLOCK_N,)), out.tile((1,)) + return x.unsqueeze(0).tile((1, -1)), y.unsqueeze(0).tile((1, -1)), out.tile((1,)) def application(x, y, out): - out = ntl.sum(x * y) + out = ntl.sum(x * y, axis=1) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_dot") diff --git a/src/ntops_lab/kernels/linear/gemm_bias.py b/src/ntops_lab/kernels/linear/gemm_bias.py index d55b179..7e99026 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias.py +++ b/src/ntops_lab/kernels/linear/gemm_bias.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_abs.py b/src/ntops_lab/kernels/linear/gemm_bias_abs.py index 8f2f3dd..a1833f9 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_abs.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_abs.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_clamp.py b/src/ntops_lab/kernels/linear/gemm_bias_clamp.py index e204c74..3ef9062 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_clamp.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_clamp.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_cube.py b/src/ntops_lab/kernels/linear/gemm_bias_cube.py index 5a92559..46f2987 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_cube.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_cube.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_elu.py b/src/ntops_lab/kernels/linear/gemm_bias_elu.py index 7c16452..6e3679e 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_elu.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_elu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_gelu.py b/src/ntops_lab/kernels/linear/gemm_bias_gelu.py index b869bbb..6f886c8 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_gelu.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_gelu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_hardshrink.py b/src/ntops_lab/kernels/linear/gemm_bias_hardshrink.py index 7c80ac8..6403883 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_hardshrink.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_hardshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,14 +13,14 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - value = acc.to(ntl.float16) + bias + value = (acc.to(ntl.float16) + bias).to(ntl.float16).to(ntl.float32) out = (ntl.where(value > 0.5, value, ntl.where(value < -0.5, value, 0.0))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_gemm_bias_hardshrink") diff --git a/src/ntops_lab/kernels/linear/gemm_bias_hardsigmoid.py b/src/ntops_lab/kernels/linear/gemm_bias_hardsigmoid.py index 6da12a2..875e777 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_hardsigmoid.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_hardsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_hardswish.py b/src/ntops_lab/kernels/linear/gemm_bias_hardswish.py index 445ff5c..75fcbab 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_hardswish.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_hardswish.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_hardtanh.py b/src/ntops_lab/kernels/linear/gemm_bias_hardtanh.py index 36eed63..86a2107 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_hardtanh.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_hardtanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_leaky_relu.py b/src/ntops_lab/kernels/linear/gemm_bias_leaky_relu.py index c3fe409..1ca6ffd 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_leaky_relu.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_leaky_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_logsigmoid.py b/src/ntops_lab/kernels/linear/gemm_bias_logsigmoid.py index 71e17c9..a915e9d 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_logsigmoid.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_logsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_neg.py b/src/ntops_lab/kernels/linear/gemm_bias_neg.py index b23cd39..366b911 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_neg.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_neg.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_relu.py b/src/ntops_lab/kernels/linear/gemm_bias_relu.py index 8454a49..345acb2 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_relu.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_relu6.py b/src/ntops_lab/kernels/linear/gemm_bias_relu6.py index f73adc4..053b96e 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_relu6.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_relu6.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_sigmoid.py b/src/ntops_lab/kernels/linear/gemm_bias_sigmoid.py index 85428fd..8c38a33 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_sigmoid.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_sigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): @@ -21,7 +21,7 @@ def application(a, b, bias, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = acc + bias - out = (ntl.sigmoid(value)).to(ntl.float16) + out = ((1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_gemm_bias_sigmoid") diff --git a/src/ntops_lab/kernels/linear/gemm_bias_silu.py b/src/ntops_lab/kernels/linear/gemm_bias_silu.py index 859c38f..fb7b518 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_silu.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_silu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): @@ -21,7 +21,7 @@ def application(a, b, bias, out): for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) value = acc + bias - out = (value * ntl.sigmoid(value)).to(ntl.float16) + out = (value * (1.0 / (1.0 + ntl.exp(-(value))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(1), Tensor(2)), kernel_name="ntops_lab_gemm_bias_silu") diff --git a/src/ntops_lab/kernels/linear/gemm_bias_softplus.py b/src/ntops_lab/kernels/linear/gemm_bias_softplus.py index 4f9a21d..737fc7f 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_softplus.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_softplus.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_softshrink.py b/src/ntops_lab/kernels/linear/gemm_bias_softshrink.py index bb13ca9..c053c2a 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_softshrink.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_softshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_square.py b/src/ntops_lab/kernels/linear/gemm_bias_square.py index 76ebb08..bdae301 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_square.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_square.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_tanh.py b/src/ntops_lab/kernels/linear/gemm_bias_tanh.py index bcee81e..d58ea52 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_tanh.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_tanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/gemm_bias_tanhshrink.py b/src/ntops_lab/kernels/linear/gemm_bias_tanhshrink.py index a339571..4dc3237 100644 --- a/src/ntops_lab/kernels/linear/gemm_bias_tanhshrink.py +++ b/src/ntops_lab/kernels/linear/gemm_bias_tanhshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/grouped_mm.py b/src/ntops_lab/kernels/linear/grouped_mm.py index ca2ae5b..105f9f2 100644 --- a/src/ntops_lab/kernels/linear/grouped_mm.py +++ b/src/ntops_lab/kernels/linear/grouped_mm.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/linear.py b/src/ntops_lab/kernels/linear/linear.py index 4c43166..abf1989 100644 --- a/src/ntops_lab/kernels/linear/linear.py +++ b/src/ntops_lab/kernels/linear/linear.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, bias, out): out_arr = out.tile((BM, BN)) @@ -13,7 +13,7 @@ def arrangement(a, b, bias, out): a_arr.dtype = a_arr.dtype.squeeze(0) b_arr = b.tile((BK, BN)).tile((-1, 1)).expand((out_arr.shape[0], -1)) b_arr.dtype = b_arr.dtype.squeeze(1) - bias_arr = bias.tile((BN,)).unsqueeze(0).expand((out_arr.shape[0], -1)) + bias_arr = bias[None, :].expand((out.shape[0], -1)).tile((BM, BN)) return a_arr, b_arr, bias_arr, out_arr def application(a, b, bias, out): diff --git a/src/ntops_lab/kernels/linear/mm.py b/src/ntops_lab/kernels/linear/mm.py index 659ec33..6e7b144 100644 --- a/src/ntops_lab/kernels/linear/mm.py +++ b/src/ntops_lab/kernels/linear/mm.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_abs.py b/src/ntops_lab/kernels/linear/mm_abs.py index a3a2754..7d2352f 100644 --- a/src/ntops_lab/kernels/linear/mm_abs.py +++ b/src/ntops_lab/kernels/linear/mm_abs.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_clamp.py b/src/ntops_lab/kernels/linear/mm_clamp.py index 7bb3fe1..1d87a7d 100644 --- a/src/ntops_lab/kernels/linear/mm_clamp.py +++ b/src/ntops_lab/kernels/linear/mm_clamp.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_cube.py b/src/ntops_lab/kernels/linear/mm_cube.py index 3ce83db..a36ba6a 100644 --- a/src/ntops_lab/kernels/linear/mm_cube.py +++ b/src/ntops_lab/kernels/linear/mm_cube.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_elu.py b/src/ntops_lab/kernels/linear/mm_elu.py index cf18958..b70c52e 100644 --- a/src/ntops_lab/kernels/linear/mm_elu.py +++ b/src/ntops_lab/kernels/linear/mm_elu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_gelu.py b/src/ntops_lab/kernels/linear/mm_gelu.py index bb972d4..2665cbf 100644 --- a/src/ntops_lab/kernels/linear/mm_gelu.py +++ b/src/ntops_lab/kernels/linear/mm_gelu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_hardshrink.py b/src/ntops_lab/kernels/linear/mm_hardshrink.py index cf7862a..8b08dae 100644 --- a/src/ntops_lab/kernels/linear/mm_hardshrink.py +++ b/src/ntops_lab/kernels/linear/mm_hardshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) @@ -19,7 +19,7 @@ def application(a, b, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - value = acc.to(ntl.float16) + value = acc.to(ntl.float16).to(ntl.float32) out = (ntl.where(value > 0.5, value, ntl.where(value < -0.5, value, 0.0))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_mm_hardshrink") diff --git a/src/ntops_lab/kernels/linear/mm_hardsigmoid.py b/src/ntops_lab/kernels/linear/mm_hardsigmoid.py index 226eb77..ec61ce8 100644 --- a/src/ntops_lab/kernels/linear/mm_hardsigmoid.py +++ b/src/ntops_lab/kernels/linear/mm_hardsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_hardswish.py b/src/ntops_lab/kernels/linear/mm_hardswish.py index 7bea909..047aa09 100644 --- a/src/ntops_lab/kernels/linear/mm_hardswish.py +++ b/src/ntops_lab/kernels/linear/mm_hardswish.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_hardtanh.py b/src/ntops_lab/kernels/linear/mm_hardtanh.py index 3f25386..9c0d061 100644 --- a/src/ntops_lab/kernels/linear/mm_hardtanh.py +++ b/src/ntops_lab/kernels/linear/mm_hardtanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_leaky_relu.py b/src/ntops_lab/kernels/linear/mm_leaky_relu.py index 443dd06..5293263 100644 --- a/src/ntops_lab/kernels/linear/mm_leaky_relu.py +++ b/src/ntops_lab/kernels/linear/mm_leaky_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_logsigmoid.py b/src/ntops_lab/kernels/linear/mm_logsigmoid.py index f374e0f..c933e4e 100644 --- a/src/ntops_lab/kernels/linear/mm_logsigmoid.py +++ b/src/ntops_lab/kernels/linear/mm_logsigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_neg.py b/src/ntops_lab/kernels/linear/mm_neg.py index 8259552..28791a0 100644 --- a/src/ntops_lab/kernels/linear/mm_neg.py +++ b/src/ntops_lab/kernels/linear/mm_neg.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_out.py b/src/ntops_lab/kernels/linear/mm_out.py index f8176bb..6f6f295 100644 --- a/src/ntops_lab/kernels/linear/mm_out.py +++ b/src/ntops_lab/kernels/linear/mm_out.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_relu.py b/src/ntops_lab/kernels/linear/mm_relu.py index 244c197..41ed6ae 100644 --- a/src/ntops_lab/kernels/linear/mm_relu.py +++ b/src/ntops_lab/kernels/linear/mm_relu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_relu6.py b/src/ntops_lab/kernels/linear/mm_relu6.py index bdbb4e6..96a7bf6 100644 --- a/src/ntops_lab/kernels/linear/mm_relu6.py +++ b/src/ntops_lab/kernels/linear/mm_relu6.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_sigmoid.py b/src/ntops_lab/kernels/linear/mm_sigmoid.py index dd5b87b..3bbe9a9 100644 --- a/src/ntops_lab/kernels/linear/mm_sigmoid.py +++ b/src/ntops_lab/kernels/linear/mm_sigmoid.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) @@ -23,7 +23,7 @@ def application(a, b, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - out = (ntl.sigmoid(acc)).to(ntl.float16) + out = ((1.0 / (1.0 + ntl.exp(-(acc))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_mm_sigmoid") diff --git a/src/ntops_lab/kernels/linear/mm_silu.py b/src/ntops_lab/kernels/linear/mm_silu.py index 40ebd2e..46d340e 100644 --- a/src/ntops_lab/kernels/linear/mm_silu.py +++ b/src/ntops_lab/kernels/linear/mm_silu.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) @@ -23,7 +23,7 @@ def application(a, b, out): acc = ntl.zeros(out.shape, dtype=ntl.float32) for k in range(a.shape[0]): acc += ntl.dot(a[k], b[k]) - out = (acc * ntl.sigmoid(acc)).to(ntl.float16) + out = (acc * (1.0 / (1.0 + ntl.exp(-(acc))))).to(ntl.float16) kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2), Tensor(2)), kernel_name="ntops_lab_mm_silu") diff --git a/src/ntops_lab/kernels/linear/mm_softplus.py b/src/ntops_lab/kernels/linear/mm_softplus.py index e09cd16..24f9239 100644 --- a/src/ntops_lab/kernels/linear/mm_softplus.py +++ b/src/ntops_lab/kernels/linear/mm_softplus.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_softshrink.py b/src/ntops_lab/kernels/linear/mm_softshrink.py index f432af0..69910e4 100644 --- a/src/ntops_lab/kernels/linear/mm_softshrink.py +++ b/src/ntops_lab/kernels/linear/mm_softshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_square.py b/src/ntops_lab/kernels/linear/mm_square.py index 416b55f..a120ce0 100644 --- a/src/ntops_lab/kernels/linear/mm_square.py +++ b/src/ntops_lab/kernels/linear/mm_square.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_tanh.py b/src/ntops_lab/kernels/linear/mm_tanh.py index af98b54..cba7101 100644 --- a/src/ntops_lab/kernels/linear/mm_tanh.py +++ b/src/ntops_lab/kernels/linear/mm_tanh.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def _arrange_matmul(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/mm_tanhshrink.py b/src/ntops_lab/kernels/linear/mm_tanhshrink.py index 7328764..f8d7558 100644 --- a/src/ntops_lab/kernels/linear/mm_tanhshrink.py +++ b/src/ntops_lab/kernels/linear/mm_tanhshrink.py @@ -3,9 +3,9 @@ import ninetoothed.language as ntl from ninetoothed import Tensor, block_size -BM = block_size() -BN = block_size() -BK = block_size() +BM = block_size(upper_bound=64) +BN = block_size(upper_bound=64) +BK = block_size(upper_bound=64) def arrangement(a, b, out): out_arr = out.tile((BM, BN)) diff --git a/src/ntops_lab/kernels/linear/vdot.py b/src/ntops_lab/kernels/linear/vdot.py index 51a53ce..30601d1 100644 --- a/src/ntops_lab/kernels/linear/vdot.py +++ b/src/ntops_lab/kernels/linear/vdot.py @@ -1,15 +1,14 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor -BLOCK_N = block_size() def arrangement(x, y, out): - return x.tile((BLOCK_N,)), y.tile((BLOCK_N,)), out.tile((1,)) + return x.unsqueeze(0).tile((1, -1)), y.unsqueeze(0).tile((1, -1)), out.tile((1,)) def application(x, y, out): - out = ntl.sum(x * y) + out = ntl.sum(x * y, axis=1) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_vdot") diff --git a/src/ntops_lab/kernels/normalization/batch_norm.py b/src/ntops_lab/kernels/normalization/batch_norm.py index a8c982b..5178ddc 100644 --- a/src/ntops_lab/kernels/normalization/batch_norm.py +++ b/src/ntops_lab/kernels/normalization/batch_norm.py @@ -1,18 +1,17 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, running_mean, running_var, weight, bias, out): - x_arr = x.tile((BLOCK_M, BLOCK_N)) - mean_arr = running_mean.tile((BLOCK_N,)).expand((x_arr.shape[0], -1)) - var_arr = running_var.tile((BLOCK_N,)).expand((x_arr.shape[0], -1)) - weight_arr = weight.tile((BLOCK_N,)).expand((x_arr.shape[0], -1)) - bias_arr = bias.tile((BLOCK_N,)).expand((x_arr.shape[0], -1)) - return x_arr, mean_arr, var_arr, weight_arr, bias_arr, out.tile((BLOCK_M, BLOCK_N)) + x_arr = x.tile((BLOCK_M, -1)) + mean_arr = running_mean.tile((-1,)).expand((x_arr.shape[0], -1)) + var_arr = running_var.tile((-1,)).expand((x_arr.shape[0], -1)) + weight_arr = weight.tile((-1,)).expand((x_arr.shape[0], -1)) + bias_arr = bias.tile((-1,)).expand((x_arr.shape[0], -1)) + return x_arr, mean_arr, var_arr, weight_arr, bias_arr, out.tile((BLOCK_M, -1)) def application(x, running_mean, running_var, weight, bias, out): out = (x - running_mean) * ntl.rsqrt(running_var + 1.0e-5) * weight + bias diff --git a/src/ntops_lab/kernels/normalization/layer_norm.py b/src/ntops_lab/kernels/normalization/layer_norm.py index 0db5796..1b6cfbf 100644 --- a/src/ntops_lab/kernels/normalization/layer_norm.py +++ b/src/ntops_lab/kernels/normalization/layer_norm.py @@ -3,24 +3,24 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, weight, bias, out, hidden): - x_arr = x.tile((BLOCK_M, BLOCK_N)) - w_arr = weight.tile((BLOCK_N,)) + x_arr = x.tile((BLOCK_M, hidden.value)) + w_arr = weight.tile((hidden.value,)) w_arr = w_arr.expand((x_arr.shape[0], -1)) - b_arr = bias.tile((BLOCK_N,)) + b_arr = bias.tile((hidden.value,)) b_arr = b_arr.expand((x_arr.shape[0], -1)) - return x_arr, w_arr, b_arr, out.tile((BLOCK_M, BLOCK_N)), hidden + return x_arr, w_arr, b_arr, out.tile((BLOCK_M, hidden.value)), hidden def application(x, weight, bias, out, hidden): - mean = ntl.sum(x, axis=1) / hidden - mean_square = ntl.sum(x * x, axis=1) / hidden - var = mean_square - mean * mean - out = (x - mean[:, None]) * ntl.rsqrt(var[:, None] + 1.0e-5) * weight + bias + value = x.to(ntl.float32) + mean = ntl.sum(value, axis=1) / hidden + centered = value - mean[:, None] + var = ntl.sum(centered * centered, axis=1) / hidden + out = (value - mean[:, None]) * ntl.rsqrt(var[:, None] + 1.0e-5) * weight + bias @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/normalization/rms_norm.py b/src/ntops_lab/kernels/normalization/rms_norm.py index 3b27c40..5f9e20b 100644 --- a/src/ntops_lab/kernels/normalization/rms_norm.py +++ b/src/ntops_lab/kernels/normalization/rms_norm.py @@ -3,20 +3,20 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, weight, out, hidden): - x_arr = x.tile((BLOCK_M, BLOCK_N)) - w_arr = weight.tile((BLOCK_N,)) + x_arr = x.tile((BLOCK_M, hidden.value)) + w_arr = weight.tile((hidden.value,)) w_arr = w_arr.expand((x_arr.shape[0], -1)) - return x_arr, w_arr, out.tile((BLOCK_M, BLOCK_N)), hidden + return x_arr, w_arr, out.tile((BLOCK_M, hidden.value)), hidden def application(x, weight, out, hidden): - mean_square = ntl.sum(x * x, axis=1) / hidden - out = x * ntl.rsqrt(mean_square[:, None] + 1.0e-5) * weight + value = x.to(ntl.float32) + mean_square = ntl.sum(value * value, axis=1) / hidden + out = value * ntl.rsqrt(mean_square[:, None] + 1.0e-5) * weight @functools.cache def _kernel(hidden): diff --git a/src/ntops_lab/kernels/pointwise/floor_divide.py b/src/ntops_lab/kernels/pointwise/floor_divide.py index 8df45e9..d28c319 100644 --- a/src/ntops_lab/kernels/pointwise/floor_divide.py +++ b/src/ntops_lab/kernels/pointwise/floor_divide.py @@ -8,7 +8,18 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = ntl.floor(x / y) + # Derive the quotient from the remainder before rounding near integers. + remainder = ntl.libdevice.fmod(x, y) + quotient = (x - remainder) / y + quotient = ntl.where( + (remainder != 0.0) & ((y < 0.0) != (remainder < 0.0)), + quotient - 1.0, + quotient, + ) + rounded = ntl.floor(quotient) + rounded = ntl.where(quotient - rounded > 0.5, rounded + 1.0, rounded) + signed_zero = ntl.libdevice.copysign(0.0, x / y) + out = ntl.where(y == 0.0, x / y, ntl.where(quotient == 0.0, signed_zero, rounded)) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="fg_split_floor_divide") diff --git a/src/ntops_lab/kernels/pointwise/floor_divide_inplace.py b/src/ntops_lab/kernels/pointwise/floor_divide_inplace.py index 2abd7d3..9834653 100644 --- a/src/ntops_lab/kernels/pointwise/floor_divide_inplace.py +++ b/src/ntops_lab/kernels/pointwise/floor_divide_inplace.py @@ -8,7 +8,18 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = ntl.floor(x / y) + # Derive the quotient from the remainder before rounding near integers. + remainder = ntl.libdevice.fmod(x, y) + quotient = (x - remainder) / y + quotient = ntl.where( + (remainder != 0.0) & ((y < 0.0) != (remainder < 0.0)), + quotient - 1.0, + quotient, + ) + rounded = ntl.floor(quotient) + rounded = ntl.where(quotient - rounded > 0.5, rounded + 1.0, rounded) + signed_zero = ntl.libdevice.copysign(0.0, x / y) + out = ntl.where(y == 0.0, x / y, ntl.where(quotient == 0.0, signed_zero, rounded)) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="fg_split_floor_divide_") diff --git a/src/ntops_lab/kernels/pointwise/floor_divide_out.py b/src/ntops_lab/kernels/pointwise/floor_divide_out.py index d2d4868..6ccdffd 100644 --- a/src/ntops_lab/kernels/pointwise/floor_divide_out.py +++ b/src/ntops_lab/kernels/pointwise/floor_divide_out.py @@ -9,7 +9,18 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = ntl.floor(x / y) + # Derive the quotient from the remainder before rounding near integers. + remainder = ntl.libdevice.fmod(x, y) + quotient = (x - remainder) / y + quotient = ntl.where( + (remainder != 0.0) & ((y < 0.0) != (remainder < 0.0)), + quotient - 1.0, + quotient, + ) + rounded = ntl.floor(quotient) + rounded = ntl.where(quotient - rounded > 0.5, rounded + 1.0, rounded) + signed_zero = ntl.libdevice.copysign(0.0, x / y) + out = ntl.where(y == 0.0, x / y, ntl.where(quotient == 0.0, signed_zero, rounded)) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_floor_divide_out") diff --git a/src/ntops_lab/kernels/pointwise/floor_divide_scalar.py b/src/ntops_lab/kernels/pointwise/floor_divide_scalar.py index 2b2bf22..30eb705 100644 --- a/src/ntops_lab/kernels/pointwise/floor_divide_scalar.py +++ b/src/ntops_lab/kernels/pointwise/floor_divide_scalar.py @@ -9,7 +9,18 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = ntl.floor(x / 1.25) + # Derive the quotient from the remainder before rounding near integers. + remainder = ntl.libdevice.fmod(x, 1.25) + quotient = (x - remainder) / 1.25 + quotient = ntl.where( + (remainder != 0.0) & ((1.25 < 0.0) != (remainder < 0.0)), + quotient - 1.0, + quotient, + ) + rounded = ntl.floor(quotient) + rounded = ntl.where(quotient - rounded > 0.5, rounded + 1.0, rounded) + signed_zero = ntl.libdevice.copysign(0.0, x / 1.25) + out = ntl.where(1.25 == 0.0, x / 1.25, ntl.where(quotient == 0.0, signed_zero, rounded)) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_floor_divide_scalar") diff --git a/src/ntops_lab/kernels/pointwise/floor_divide_scalar_out.py b/src/ntops_lab/kernels/pointwise/floor_divide_scalar_out.py index f3d8146..74fd017 100644 --- a/src/ntops_lab/kernels/pointwise/floor_divide_scalar_out.py +++ b/src/ntops_lab/kernels/pointwise/floor_divide_scalar_out.py @@ -9,7 +9,18 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = ntl.floor(x / 1.25) + # Derive the quotient from the remainder before rounding near integers. + remainder = ntl.libdevice.fmod(x, 1.25) + quotient = (x - remainder) / 1.25 + quotient = ntl.where( + (remainder != 0.0) & ((1.25 < 0.0) != (remainder < 0.0)), + quotient - 1.0, + quotient, + ) + rounded = ntl.floor(quotient) + rounded = ntl.where(quotient - rounded > 0.5, rounded + 1.0, rounded) + signed_zero = ntl.libdevice.copysign(0.0, x / 1.25) + out = ntl.where(1.25 == 0.0, x / 1.25, ntl.where(quotient == 0.0, signed_zero, rounded)) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_floor_divide_scalar_out") diff --git a/src/ntops_lab/kernels/pointwise/fmod.py b/src/ntops_lab/kernels/pointwise/fmod.py index 2f8e08b..f6ea93a 100644 --- a/src/ntops_lab/kernels/pointwise/fmod.py +++ b/src/ntops_lab/kernels/pointwise/fmod.py @@ -9,7 +9,7 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.where((x / y) < 0.0, ntl.ceil(x / y), ntl.floor(x / y)) * y + out = ntl.libdevice.fmod(x, y) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_fmod") diff --git a/src/ntops_lab/kernels/pointwise/fmod_out.py b/src/ntops_lab/kernels/pointwise/fmod_out.py index 9864870..c6d491e 100644 --- a/src/ntops_lab/kernels/pointwise/fmod_out.py +++ b/src/ntops_lab/kernels/pointwise/fmod_out.py @@ -9,7 +9,7 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.floor(x / y) * y + out = ntl.libdevice.fmod(x, y) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_fmod_out") diff --git a/src/ntops_lab/kernels/pointwise/fmod_scalar.py b/src/ntops_lab/kernels/pointwise/fmod_scalar.py index 2c9cef8..36ff664 100644 --- a/src/ntops_lab/kernels/pointwise/fmod_scalar.py +++ b/src/ntops_lab/kernels/pointwise/fmod_scalar.py @@ -9,7 +9,7 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.floor(x / 1.25) * 1.25 + out = ntl.libdevice.fmod(x, 1.25) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_fmod_scalar") diff --git a/src/ntops_lab/kernels/pointwise/fmod_scalar_out.py b/src/ntops_lab/kernels/pointwise/fmod_scalar_out.py index 60b2603..b143686 100644 --- a/src/ntops_lab/kernels/pointwise/fmod_scalar_out.py +++ b/src/ntops_lab/kernels/pointwise/fmod_scalar_out.py @@ -9,7 +9,7 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.floor(x / 1.25) * 1.25 + out = ntl.libdevice.fmod(x, 1.25) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_fmod_scalar_out") diff --git a/src/ntops_lab/kernels/pointwise/functional_sigmoid.py b/src/ntops_lab/kernels/pointwise/functional_sigmoid.py index 72b4058..7214d28 100644 --- a/src/ntops_lab/kernels/pointwise/functional_sigmoid.py +++ b/src/ntops_lab/kernels/pointwise/functional_sigmoid.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.sigmoid(x) + out = (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_functional_sigmoid") diff --git a/src/ntops_lab/kernels/pointwise/functional_sigmoid_out.py b/src/ntops_lab/kernels/pointwise/functional_sigmoid_out.py index c12566f..5ca95ea 100644 --- a/src/ntops_lab/kernels/pointwise/functional_sigmoid_out.py +++ b/src/ntops_lab/kernels/pointwise/functional_sigmoid_out.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.sigmoid(x) + out = (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_functional_sigmoid_out") diff --git a/src/ntops_lab/kernels/pointwise/functional_silu.py b/src/ntops_lab/kernels/pointwise/functional_silu.py index f615e4c..a3fb783 100644 --- a/src/ntops_lab/kernels/pointwise/functional_silu.py +++ b/src/ntops_lab/kernels/pointwise/functional_silu.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = x * ntl.sigmoid(x) + out = x * (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_functional_silu") diff --git a/src/ntops_lab/kernels/pointwise/functional_silu_out.py b/src/ntops_lab/kernels/pointwise/functional_silu_out.py index 977292f..465de92 100644 --- a/src/ntops_lab/kernels/pointwise/functional_silu_out.py +++ b/src/ntops_lab/kernels/pointwise/functional_silu_out.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = x * ntl.sigmoid(x) + out = x * (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_functional_silu_out") diff --git a/src/ntops_lab/kernels/pointwise/isfinite.py b/src/ntops_lab/kernels/pointwise/isfinite.py index 5139b17..b5af4b1 100644 --- a/src/ntops_lab/kernels/pointwise/isfinite.py +++ b/src/ntops_lab/kernels/pointwise/isfinite.py @@ -1,5 +1,6 @@ import torch import ninetoothed +import ninetoothed.language as ntl from ninetoothed import Tensor, block_size @@ -9,7 +10,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = (x == x) & (x != float("inf")) & (x != -float("inf")) + out = (ntl.libdevice.isnan(x) == 0) & (x != float("inf")) & (x != -float("inf")) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_isfinite") diff --git a/src/ntops_lab/kernels/pointwise/isnan.py b/src/ntops_lab/kernels/pointwise/isnan.py index ac1ad89..2d55e2f 100644 --- a/src/ntops_lab/kernels/pointwise/isnan.py +++ b/src/ntops_lab/kernels/pointwise/isnan.py @@ -1,5 +1,6 @@ import torch import ninetoothed +import ninetoothed.language as ntl from ninetoothed import Tensor, block_size @@ -9,7 +10,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = x != x + out = ntl.libdevice.isnan(x) != 0 kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_isnan") diff --git a/src/ntops_lab/kernels/pointwise/remainder.py b/src/ntops_lab/kernels/pointwise/remainder.py index 8208df9..9f90f64 100644 --- a/src/ntops_lab/kernels/pointwise/remainder.py +++ b/src/ntops_lab/kernels/pointwise/remainder.py @@ -9,7 +9,8 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - y * ntl.floor(x / y) + r = ntl.where(ntl.abs(x) < ntl.abs(y), x, ntl.libdevice.fmod(x, y)) + out = ntl.where((r != 0.0) & ((r < 0.0) != (y < 0.0)), r + y, r) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_remainder") diff --git a/src/ntops_lab/kernels/pointwise/remainder_inplace.py b/src/ntops_lab/kernels/pointwise/remainder_inplace.py index af0f20a..db0ef87 100644 --- a/src/ntops_lab/kernels/pointwise/remainder_inplace.py +++ b/src/ntops_lab/kernels/pointwise/remainder_inplace.py @@ -8,7 +8,8 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - y * ntl.floor(x / y) + r = ntl.where(ntl.abs(x) < ntl.abs(y), x, ntl.libdevice.fmod(x, y)) + out = ntl.where((r != 0.0) & ((r < 0.0) != (y < 0.0)), r + y, r) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="fg_split_remainder_") diff --git a/src/ntops_lab/kernels/pointwise/remainder_out.py b/src/ntops_lab/kernels/pointwise/remainder_out.py index 561bd17..461cd41 100644 --- a/src/ntops_lab/kernels/pointwise/remainder_out.py +++ b/src/ntops_lab/kernels/pointwise/remainder_out.py @@ -9,7 +9,8 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.floor(x / y) * y + r = ntl.where(ntl.abs(x) < ntl.abs(y), x, ntl.libdevice.fmod(x, y)) + out = ntl.where((r != 0.0) & ((r < 0.0) != (y < 0.0)), r + y, r) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_remainder_out") diff --git a/src/ntops_lab/kernels/pointwise/remainder_scalar.py b/src/ntops_lab/kernels/pointwise/remainder_scalar.py index 31ae44b..5ed288e 100644 --- a/src/ntops_lab/kernels/pointwise/remainder_scalar.py +++ b/src/ntops_lab/kernels/pointwise/remainder_scalar.py @@ -9,7 +9,8 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.floor(x / 1.25) * 1.25 + r = ntl.where(ntl.abs(x) < ntl.abs(1.25), x, ntl.libdevice.fmod(x, 1.25)) + out = ntl.where((r != 0.0) & ((r < 0.0) != (1.25 < 0.0)), r + 1.25, r) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_remainder_scalar") diff --git a/src/ntops_lab/kernels/pointwise/remainder_scalar_out.py b/src/ntops_lab/kernels/pointwise/remainder_scalar_out.py index 7d33e93..893cdfc 100644 --- a/src/ntops_lab/kernels/pointwise/remainder_scalar_out.py +++ b/src/ntops_lab/kernels/pointwise/remainder_scalar_out.py @@ -9,7 +9,8 @@ def arrangement(x, y, out): return x.tile((BLOCK_SIZE,)), y.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, y, out): - out = x - ntl.floor(x / 1.25) * 1.25 + r = ntl.where(ntl.abs(x) < ntl.abs(1.25), x, ntl.libdevice.fmod(x, 1.25)) + out = ntl.where((r != 0.0) & ((r < 0.0) != (1.25 < 0.0)), r + 1.25, r) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1), Tensor(1)), kernel_name="ntops_lab_remainder_scalar_out") diff --git a/src/ntops_lab/kernels/pointwise/round.py b/src/ntops_lab/kernels/pointwise/round.py index 950c864..ad0f2a2 100644 --- a/src/ntops_lab/kernels/pointwise/round.py +++ b/src/ntops_lab/kernels/pointwise/round.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.floor(x + 0.5) + out = ntl.libdevice.nearbyint(x) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_round") diff --git a/src/ntops_lab/kernels/pointwise/round_out.py b/src/ntops_lab/kernels/pointwise/round_out.py index c6f0fb9..bf8705d 100644 --- a/src/ntops_lab/kernels/pointwise/round_out.py +++ b/src/ntops_lab/kernels/pointwise/round_out.py @@ -9,7 +9,10 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.floor(x + 0.5) + lower = ntl.floor(x) + fraction = x - lower + odd = lower - 2.0 * ntl.floor(lower * 0.5) + out = ntl.where(fraction > 0.5, lower + 1.0, ntl.where(fraction == 0.5, lower + odd, lower)) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_round_out") diff --git a/src/ntops_lab/kernels/pointwise/sigmoid_out.py b/src/ntops_lab/kernels/pointwise/sigmoid_out.py index 1747916..ada2575 100644 --- a/src/ntops_lab/kernels/pointwise/sigmoid_out.py +++ b/src/ntops_lab/kernels/pointwise/sigmoid_out.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.sigmoid(x) + out = (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_sigmoid_out") diff --git a/src/ntops_lab/kernels/pointwise/silu_out.py b/src/ntops_lab/kernels/pointwise/silu_out.py index f3499c0..d711b43 100644 --- a/src/ntops_lab/kernels/pointwise/silu_out.py +++ b/src/ntops_lab/kernels/pointwise/silu_out.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = x * ntl.sigmoid(x) + out = x * (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_silu_out") diff --git a/src/ntops_lab/kernels/pointwise/special_expit.py b/src/ntops_lab/kernels/pointwise/special_expit.py index 9111eb2..0f3420e 100644 --- a/src/ntops_lab/kernels/pointwise/special_expit.py +++ b/src/ntops_lab/kernels/pointwise/special_expit.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.sigmoid(x) + out = (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_special_expit") diff --git a/src/ntops_lab/kernels/pointwise/special_expit_out.py b/src/ntops_lab/kernels/pointwise/special_expit_out.py index 22af1e9..53a9e48 100644 --- a/src/ntops_lab/kernels/pointwise/special_expit_out.py +++ b/src/ntops_lab/kernels/pointwise/special_expit_out.py @@ -9,7 +9,7 @@ def arrangement(x, out): return x.tile((BLOCK_SIZE,)), out.tile((BLOCK_SIZE,)) def application(x, out): - out = ntl.sigmoid(x) + out = (1.0 / (1.0 + ntl.exp(-(x)))) kernel = ninetoothed.make(arrangement, application, (Tensor(1), Tensor(1)), kernel_name="ntops_lab_special_expit_out") diff --git a/src/ntops_lab/kernels/reduction/aminmax.py b/src/ntops_lab/kernels/reduction/aminmax.py index 334ac1f..4adf044 100644 --- a/src/ntops_lab/kernels/reduction/aminmax.py +++ b/src/ntops_lab/kernels/reduction/aminmax.py @@ -1,23 +1,7 @@ -import torch -import ninetoothed -import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from .min import run as run_min +from .max import run as run_max -BM = 1 -BN = block_size() - -def arrangement(x, out0, out1): - return x.tile((BM, BN)), out0.tile((BM,)), out1.tile((BM,)) - -def application(x, out0, out1): - out0 = ntl.min(x, axis=1) - out1 = ntl.max(x, axis=1) - -kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(1), Tensor(1)), kernel_name="ntops_lab_aminmax") def run(*inputs): - x, = inputs - out0 = torch.empty((x.shape[0],), device=x.device, dtype=x.dtype) - out1 = torch.empty((x.shape[0],), device=x.device, dtype=x.dtype) - kernel(x, out0, out1) - return out0, out1 + """Compute both row reductions with independent, supported output domains.""" + return run_min(*inputs), run_max(*inputs) diff --git a/src/ntops_lab/kernels/reduction/cross_entropy.py b/src/ntops_lab/kernels/reduction/cross_entropy.py index 3c53b12..d7787be 100644 --- a/src/ntops_lab/kernels/reduction/cross_entropy.py +++ b/src/ntops_lab/kernels/reduction/cross_entropy.py @@ -12,8 +12,9 @@ def arrangement(x_for_logsumexp, x_for_gather, target_ids, class_ids, out): return x_for_logsumexp.tile((1, classes)), x_for_gather.tile((1, classes)), target_ids.tile((1, classes)), class_ids.tile((1, classes)), out.tile((1,)) def application(x_for_logsumexp, x_for_gather, target_ids, class_ids, out): - log_z = ntl.log(ntl.sum(ntl.exp(x_for_logsumexp), axis=1)) - selected = ntl.sum(x_for_gather * (target_ids == class_ids), axis=1) + maximum = ntl.max(x_for_logsumexp, axis=1) + log_z = maximum + ntl.log(ntl.sum(ntl.exp(x_for_logsumexp - maximum[:, None]), axis=1)) + selected = ntl.sum(ntl.where(target_ids == class_ids, x_for_gather, 0.0), axis=1) out = log_z - selected return ninetoothed.make( diff --git a/src/ntops_lab/kernels/reduction/log_softmax.py b/src/ntops_lab/kernels/reduction/log_softmax.py index 7a0355a..0428b26 100644 --- a/src/ntops_lab/kernels/reduction/log_softmax.py +++ b/src/ntops_lab/kernels/reduction/log_softmax.py @@ -1,20 +1,20 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M, BLOCK_N)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M, -1)) def application(x, out): - m = ntl.max(x, axis=1) - e = ntl.exp(x - m[:, None]) - out = x - m[:, None] - ntl.log(ntl.sum(e, axis=1))[:, None] + value = x.to(ntl.float32) + m = ntl.max(value, axis=1) + e = ntl.exp(value - m[:, None]) + out = value - m[:, None] - ntl.log(ntl.sum(e, axis=1))[:, None] -kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2)), kernel_name="ntops_lab_log_softmax") +kernel = ninetoothed.make(arrangement, application, (Tensor(2, other=float("-inf")), Tensor(2)), kernel_name="ntops_lab_log_softmax") def run(*inputs): x, = inputs diff --git a/src/ntops_lab/kernels/reduction/logsumexp.py b/src/ntops_lab/kernels/reduction/logsumexp.py index c9cefb3..817ae24 100644 --- a/src/ntops_lab/kernels/reduction/logsumexp.py +++ b/src/ntops_lab/kernels/reduction/logsumexp.py @@ -1,18 +1,20 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M,)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M,)) def application(x, out): - out = ntl.log(ntl.sum(ntl.exp(x), axis=1)) + value = x.to(ntl.float32) + maximum = ntl.max(value, axis=1) + shifted = value - maximum[:, None] + out = maximum + ntl.log(ntl.sum(ntl.exp(shifted), axis=1)) -kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(1)), kernel_name="ntops_lab_logsumexp") +kernel = ninetoothed.make(arrangement, application, (Tensor(2, other=float("-inf")), Tensor(1)), kernel_name="ntops_lab_logsumexp") def run(*inputs): x, = inputs diff --git a/src/ntops_lab/kernels/reduction/max.py b/src/ntops_lab/kernels/reduction/max.py index 37aca31..0d02668 100644 --- a/src/ntops_lab/kernels/reduction/max.py +++ b/src/ntops_lab/kernels/reduction/max.py @@ -1,13 +1,12 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M,)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M,)) def application(x, out): out = ntl.max(x, axis=1) diff --git a/src/ntops_lab/kernels/reduction/min.py b/src/ntops_lab/kernels/reduction/min.py index d73f5f3..048a19d 100644 --- a/src/ntops_lab/kernels/reduction/min.py +++ b/src/ntops_lab/kernels/reduction/min.py @@ -1,13 +1,12 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M,)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M,)) def application(x, out): out = ntl.min(x, axis=1) diff --git a/src/ntops_lab/kernels/reduction/safe_softmax.py b/src/ntops_lab/kernels/reduction/safe_softmax.py index ced8d4f..78d6b96 100644 --- a/src/ntops_lab/kernels/reduction/safe_softmax.py +++ b/src/ntops_lab/kernels/reduction/safe_softmax.py @@ -1,20 +1,21 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M, BLOCK_N)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M, -1)) def application(x, out): m = ntl.max(x, axis=1) - e = ntl.exp(x - m[:, None]) - out = e / ntl.sum(e, axis=1)[:, None] + shifted = ntl.where(m[:, None] == float("-inf"), float("-inf"), x - m[:, None]) + e = ntl.exp(shifted) + denom = ntl.sum(e, axis=1)[:, None] + out = e / ntl.where(denom == 0.0, 1.0, denom) -kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2)), kernel_name="ntops_lab_safe_softmax") +kernel = ninetoothed.make(arrangement, application, (Tensor(2, other=float("-inf")), Tensor(2)), kernel_name="ntops_lab_safe_softmax") def run(*inputs): x, = inputs diff --git a/src/ntops_lab/kernels/reduction/scaled_softmax.py b/src/ntops_lab/kernels/reduction/scaled_softmax.py index 2878b75..f869dbc 100644 --- a/src/ntops_lab/kernels/reduction/scaled_softmax.py +++ b/src/ntops_lab/kernels/reduction/scaled_softmax.py @@ -1,20 +1,20 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M, BLOCK_N)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M, -1)) def application(x, out): - m = ntl.max(x, axis=1) - e = ntl.exp(x - m[:, None]) + value = x.to(ntl.float32) + m = ntl.max(value, axis=1) + e = ntl.exp(value - m[:, None]) out = e / ntl.sum(e, axis=1)[:, None] -kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2)), kernel_name="ntops_lab_scaled_softmax") +kernel = ninetoothed.make(arrangement, application, (Tensor(2, other=float("-inf")), Tensor(2)), kernel_name="ntops_lab_scaled_softmax") def run(*inputs): x, = inputs diff --git a/src/ntops_lab/kernels/reduction/softmax.py b/src/ntops_lab/kernels/reduction/softmax.py index 94449d5..8a6073d 100644 --- a/src/ntops_lab/kernels/reduction/softmax.py +++ b/src/ntops_lab/kernels/reduction/softmax.py @@ -1,20 +1,20 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M, BLOCK_N)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M, -1)) def application(x, out): - m = ntl.max(x, axis=1) - e = ntl.exp(x - m[:, None]) + value = x.to(ntl.float32) + m = ntl.max(value, axis=1) + e = ntl.exp(value - m[:, None]) out = e / ntl.sum(e, axis=1)[:, None] -kernel = ninetoothed.make(arrangement, application, (Tensor(2), Tensor(2)), kernel_name="ntops_lab_softmax") +kernel = ninetoothed.make(arrangement, application, (Tensor(2, other=float("-inf")), Tensor(2)), kernel_name="ntops_lab_softmax") def run(*inputs): x, = inputs diff --git a/src/ntops_lab/kernels/reduction/sum.py b/src/ntops_lab/kernels/reduction/sum.py index 04d7022..cd6a244 100644 --- a/src/ntops_lab/kernels/reduction/sum.py +++ b/src/ntops_lab/kernels/reduction/sum.py @@ -1,13 +1,12 @@ import torch import ninetoothed import ninetoothed.language as ntl -from ninetoothed import Tensor, block_size +from ninetoothed import Tensor BLOCK_M = 1 -BLOCK_N = block_size() def arrangement(x, out): - return x.tile((BLOCK_M, BLOCK_N)), out.tile((BLOCK_M,)) + return x.tile((BLOCK_M, -1)), out.tile((BLOCK_M,)) def application(x, out): out = ntl.sum(x, axis=1) diff --git a/src/ntops_lab/kernels/reduction/var_mean.py b/src/ntops_lab/kernels/reduction/var_mean.py index 7a39819..5053f33 100644 --- a/src/ntops_lab/kernels/reduction/var_mean.py +++ b/src/ntops_lab/kernels/reduction/var_mean.py @@ -1,38 +1,7 @@ -import functools - -import torch -import ninetoothed -import ninetoothed.language as ntl -from ninetoothed import Tensor - - -def arrangement(x, out0, out1, dim): - return x.tile((1, dim.value)), out0.tile((1,)), out1.tile((1,)), dim - - -def application(x, out0, out1, dim): - mean = ntl.sum(x, axis=1) / dim - var = ntl.sum(x * x, axis=1) / dim - mean * mean - out0 = var - out1 = mean - - -@functools.cache -def _kernel(dim): - dim_tensor = Tensor(0, constexpr=True, value=dim, name="dim") - return ninetoothed.make( - arrangement, - application, - (Tensor(2), Tensor(1), Tensor(1), dim_tensor), - kernel_name=f"ntops_lab_var_mean_d{dim}", - max_num_configs=1, - ) +from .var import run as run_var +from .mean import run as run_mean def run(*inputs): - x, = inputs - out0 = torch.empty((x.shape[0],), device=x.device, dtype=x.dtype) - out1 = torch.empty((x.shape[0],), device=x.device, dtype=x.dtype) - dim = x.shape[-1] - _kernel(dim)(x, out0, out1, dim) - return out0, out1 + """Return population variance and mean, using two NineToothed kernels.""" + return run_var(*inputs), run_mean(*inputs) diff --git a/tests/test_current_compiler_regressions.py b/tests/test_current_compiler_regressions.py new file mode 100644 index 0000000..37e5b44 --- /dev/null +++ b/tests/test_current_compiler_regressions.py @@ -0,0 +1,102 @@ +import os + +import pytest + +from ntops_lab import ops + +torch = pytest.importorskip("torch") + +pytestmark = pytest.mark.skipif( + os.environ.get("NTOPS_RUN_OPERATOR_VALIDATION") != "1", + reason="set NTOPS_RUN_OPERATOR_VALIDATION=1 to run GPU regressions", +) + + +@pytest.fixture(autouse=True) +def seed(): + torch.manual_seed(20260917) + + +@pytest.mark.parametrize("width", [1, 7, 33, 257]) +@pytest.mark.parametrize("dtype", [torch.float32, torch.float16, torch.bfloat16]) +def test_softmax_row_domain(width, dtype): + x = torch.randn((3, width), device="cuda", dtype=dtype) + tol = 1e-5 if dtype == torch.float32 else 0.01 + torch.testing.assert_close(ops.softmax(x), x.softmax(-1), atol=tol, rtol=tol) + torch.testing.assert_close( + ops.log_softmax(x), x.log_softmax(-1), atol=tol, rtol=tol + ) + + +def test_safe_softmax_fully_masked_rows(): + x = torch.randn((3, 33), device="cuda") + x[0] = float("-inf") + expected = x.softmax(-1) + expected[0] = 0 + torch.testing.assert_close(ops.get_op("_safe_softmax")(x), expected) + + +@pytest.mark.parametrize("width", [7, 33, 257]) +@pytest.mark.parametrize("dtype", [torch.float32, torch.float16, torch.bfloat16]) +def test_normalization_full_hidden_axis(width, dtype): + x = torch.randn((3, width), device="cuda", dtype=dtype) + w = torch.randn(width, device="cuda", dtype=dtype) + b = torch.randn_like(w) + tol = 2e-5 if dtype == torch.float32 else 0.02 + expected = torch.nn.functional.layer_norm(x, (width,), w, b, 1e-5) + torch.testing.assert_close(ops.layer_norm(x, w, b), expected, atol=tol, rtol=tol) + expected_rms = ( + x.float() + * torch.rsqrt(x.float().square().mean(-1, keepdim=True) + 1e-5) + * w.float() + ).to(dtype) + torch.testing.assert_close(ops.rms_norm(x, w), expected_rms, atol=tol, rtol=tol) + + +def test_layer_norm_large_offset(): + x = torch.randn((3, 33), device="cuda") + 10000 + w = torch.ones(33, device="cuda") + b = torch.zeros_like(w) + expected = torch.nn.functional.layer_norm(x, (33,), w, b, 1e-5) + torch.testing.assert_close(ops.layer_norm(x, w, b), expected, atol=0.01, rtol=0.01) + + +@pytest.mark.parametrize("width", [7, 33, 257]) +def test_full_axis_and_multiple_reductions(width): + x = torch.randn((3, width), device="cuda") + torch.testing.assert_close(ops.sum(x), x.sum(-1)) + actual_min, actual_max = ops.aminmax(x) + torch.testing.assert_close(actual_min, x.amin(-1)) + torch.testing.assert_close(actual_max, x.amax(-1)) + actual_var, actual_mean = ops.var_mean(x) + torch.testing.assert_close(actual_var, x.var(-1, correction=0)) + torch.testing.assert_close(actual_mean, x.mean(-1)) + y = torch.randn(width, device="cuda") + torch.testing.assert_close(ops.dot(x[0], y), torch.dot(x[0], y).reshape(1)) + + +@pytest.mark.parametrize("offset", [-100, 0, 100]) +def test_cross_entropy_stability(offset): + x = torch.randn((3, 33), device="cuda") + offset + target = torch.tensor([0, 7, 32], device="cuda") + expected = torch.nn.functional.cross_entropy(x, target, reduction="none") + torch.testing.assert_close( + ops.cross_entropy(x, target), expected, atol=2e-5, rtol=2e-5 + ) + + +def test_round_out_ties_to_even(): + x = torch.tensor( + [-2.5, -1.5, -0.5, 0.5, 1.5, 2.5, float("inf"), float("-inf"), float("nan")], + device="cuda", + ) + torch.testing.assert_close(ops.round_out(x), x.round(), equal_nan=True) + + +def test_floor_divide_integer_boundaries(): + x = torch.tensor([1.0, -1.0, 0.6, -0.6, 0.0, -0.0, 1.0], device="cuda") + y = torch.tensor([0.1, 0.1, 0.1, 0.1, 2.0, 2.0, 0.0], device="cuda") + actual = ops.floor_divide_out(x, y) + expected = torch.floor_divide(x, y) + torch.testing.assert_close(actual, expected, atol=0, rtol=0) + assert torch.equal(torch.signbit(actual), torch.signbit(expected))