Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 14 additions & 1 deletion numba/cuda/mathimpl.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import math
import operator
from llvmlite import ir
from numba.core import types, typing, utils
from numba.core import types, typing, utils, cgutils
from numba.core.imputils import Registry
from numba.types import float32, float64, int64, uint64
from numba.cuda import libdevice
Expand Down Expand Up @@ -71,6 +71,19 @@ def math_isinf_isnan_int(context, builder, sig, args):
return context.get_constant(types.boolean, 0)


@lower(operator.truediv, types.float32, types.float32)
def maybe_fast_truediv(context, builder, sig, args):
if context.fastmath:
sig = typing.signature(float32, float32, float32)
impl = context.get_function(libdevice.fast_fdividef, sig)
return impl(builder, args)
else:
with cgutils.if_zero(builder, args[1]):
context.error_model.fp_zero_division(builder, ("division by zero",))
res = builder.fdiv(*args)
return res


@lower(math.isfinite, types.Integer)
def math_isfinite_int(context, builder, sig, args):
return context.get_constant(types.boolean, 1)
Expand Down
40 changes: 37 additions & 3 deletions numba/cuda/tests/cudapy/test_fastmath.py
Original file line number Diff line number Diff line change
@@ -1,12 +1,14 @@
from numba import cuda, float32
from math import cos, sin, tan, exp, log, log10, log2, pow
import numpy as np
from numba.cuda.testing import CUDATestCase, skip_on_cudasim
import unittest


class TestFastMathOption(CUDATestCase):
@skip_on_cudasim('fast divide not available in CUDASIM')
def test_kernel(self):

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

For a while I was wondering what this is checking that differs from what test_divf checks for below... Eventually I concluded that it tests that casting an integer to a float32 then using it in a division results in a div.approx instruction being emitted - given how finicky casts and the type system can be, I think it's good to keep this check alongside test_divf - Could you add a comment here explaining that this is testing the cast of an int being used in the divide, to make it a bit easier for future readers to see the difference?

Alternatively - do you have a different view on the difference between this and test_divf? I reached my conclusion by trying to derive meaning from the code (thinking backwards a bit) so maybe there's something else going on here that I haven't followed.


# Test the cast of an int being used in fastmath divide
def foo(arr, val):
i = cuda.grid(1)
if i < arr.size:
Expand All @@ -15,8 +17,8 @@ def foo(arr, val):
fastver = cuda.jit("void(float32[:], float32)", fastmath=True)(foo)
precver = cuda.jit("void(float32[:], float32)")(foo)

self.assertIn('div.full.ftz.f32', fastver.ptx)
self.assertNotIn('div.full.ftz.f32', precver.ptx)
self.assertIn('div.approx.ftz.f32', fastver.ptx)
self.assertNotIn('div.approx.ftz.f32', precver.ptx)
Comment on lines +20 to +21

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In addition to replacing the check here, we could also check that the non-fastmath version uses the non-ftz version of the divide too:

Suggested change
self.assertIn('div.approx.ftz.f32', fastver.ptx)
self.assertNotIn('div.approx.ftz.f32', precver.ptx)
self.assertIn('div.approx.ftz.f32', fastver.ptx)
self.assertNotIn('div.approx.ftz.f32', precver.ptx)
self.assertIn('div.rn.f32', precver.ptx)
self.assertNotIn('div.rn.f32', fastver.ptx)


@skip_on_cudasim('fast cos not available in CUDASIM')
def test_cosf(self):
Expand Down Expand Up @@ -105,6 +107,38 @@ def f8(r, x, y):
self.assertIn('lg2.approx.ftz.f32 ', fastver.ptx)
self.assertNotIn('lg2.approx.ftz.f32 ', slowver.ptx)

@skip_on_cudasim('fast divide not available in CUDASIM')
def test_divf(self):
def f9(r, x, y):
r[0] = x / y

fastver = cuda.jit("void(float32[::1], float32, float32)",
fastmath=True)(f9)
slowver = cuda.jit("void(float32[::1], float32, float32)")(f9)
self.assertIn('div.approx.ftz.f32 ', fastver.ptx)
self.assertNotIn('div.approx.ftz.f32 ', slowver.ptx)
self.assertIn('div.rn.f32', slowver.ptx)
self.assertNotIn('div.rn.f32', fastver.ptx)

@skip_on_cudasim('fast divide not available in CUDASIM')
def test_divf_exception(self):
def f10(r, x, y):
r[0] = x / y

fastver = cuda.jit("void(float32[::1], float32, float32)",
fastmath=True, debug=True)(f10)
slowver = cuda.jit("void(float32[::1], float32, float32)",
debug=True)(f10)
nelem = 10
ary = np.empty(nelem, dtype=np.float32)
with self.assertRaises(ZeroDivisionError):
slowver[1, nelem](ary, 10.0, 0.0)

try:
fastver[1, nelem](ary, 10.0, 0.0)
except ZeroDivisionError:
self.fail("Divide in fastmath should not throw ZeroDivisionError")

def test_device(self):
# fastmath option is ignored for device function
@cuda.jit("float32(float32, float32)", device=True)
Expand Down