This commit adds explicit casts to float for -INFINITY.
The motivation for this is that in CUDA 11.8.0, the -INFINITY macro is
defined as a double (a header provided NVCC). This triggers a warning
and hence causes a CI failure in whisper.cpp. I belive that this header
might have been updated in CUDA 12 which is why we don't see this
warning.
Refs: https://github.com/ggml-org/whisper.cpp/actions/runs/
25713948217/job/
75500081939?pr=3803
Refs: https://github.com/ggml-org/llama.cpp/issues/22824
static __device__ T sentinel() {
if constexpr (std::is_same_v<T, float>) {
- return -INFINITY;
+ return -(float)INFINITY;
} else if constexpr (std::is_same_v<T, half2>) {
- return make_half2(-INFINITY, -INFINITY);
+ return make_half2(__float2half(-(float)INFINITY), __float2half(-(float)INFINITY));
} else {
static_assert(ggml_cuda_dependent_false_v<T>, "Unsupported type for block reduce max");
}