af::max returns the right maximum value but the wrong idx when the input is a subarray.
Using copy() on the subarray fixes the problem.
It happens with CUDA, OpenCL and CPU backends,
Description
- Additional details regarding the bug
- Did you build ArrayFire yourself or did you use the official installers
Official installer
- Which backend is experiencing this issue? (CPU, CUDA, OpenCL)
CPU, CUDA, OpenCL
- Do you have a workaround?
Copy the subarray
- Can the bug be reproduced reliably on your system?
yes
-->
Reproducible Code and/or Steps
`
#include <arrayfire.h>
#include
int main(int argc, char const* argv[]) {
try {
af::setBackend(AF_BACKEND_CUDA);
af::setDevice(0);
af::info();
float input_vals[25] = {0.0168, 0.0278, 0.0317, 0.0248, 0.0131, 0.0197, 0.0321, 0.0362, 0.0279,
0.0141, 0.0218, 0.0353, 0.0394, 0.0297, 0.0143, 0.0224, 0.0363, 0.0404,
0.0302, 0.0142, 0.0217, 0.0355, 0.0398, 0.0302, 0.0144};
af::array input(5, 5, input_vals);
af::print("input", input);
unsigned idx;
float max_val;
af::max<float>(&max_val, &idx, input);
std::cout << "idx: " << idx << ", max_val: " << max_val << std::endl; //OK
unsigned row = idx % input.dims(0);
unsigned col = idx / input.dims(0);
af::array input_cropped = input(af::seq(row - 1, row + 1), af::seq(col - 1, col + 1));
af::print("input_cropped", input_cropped);
af::max<float>(&max_val, &idx, input_cropped);
std::cout << "idx: " << idx << ", max_val: " << max_val << std::endl; //NOK, wrong idx (6 instead of 4), right maximum value
af::print("input_cropped.copy()", input_cropped.copy());
af::max<float>(&max_val, &idx, input_cropped.copy());
std::cout << "idx: " << idx << ", max_val: " << max_val << std::endl; //OK
} catch (const af::exception& e) {
std::cerr << e.what() << '\n';
}
return 0;
}
`
.\test.exe
ArrayFire v3.9.0 (CUDA, 64-bit Windows, build b59a1ae)
Platform: CUDA Runtime 12.2, Driver: 12080
[0] NVIDIA RTX A2000 8GB Laptop GPU, 8192 MB, CUDA Compute 8.6
input
[5 5 1 1]
0.0168 0.0197 0.0218 0.0224 0.0217
0.0278 0.0321 0.0353 0.0363 0.0355
0.0317 0.0362 0.0394 0.0404 0.0398
0.0248 0.0279 0.0297 0.0302 0.0302
0.0131 0.0141 0.0143 0.0142 0.0144
idx: 17, max_val: 0.0404
input_cropped
[3 3 1 1]
0.0353 0.0363 0.0355
0.0394 0.0404 0.0398
0.0297 0.0302 0.0302
idx: 6, max_val: 0.0404
input_cropped.copy()
[3 3 1 1]
0.0353 0.0363 0.0355
0.0394 0.0404 0.0398
0.0297 0.0302 0.0302
idx: 4, max_val: 0.0404
System Information
- ArrayFire version
3.9.0 on Windows 10
Checklist
- [ x ] Using the latest available ArrayFire release
- [ x ] GPU drivers are up to date
af::max returns the right maximum value but the wrong idx when the input is a subarray.
Using copy() on the subarray fixes the problem.
It happens with CUDA, OpenCL and CPU backends,
Description
Official installer
CPU, CUDA, OpenCL
Copy the subarray
yes
-->
Reproducible Code and/or Steps
`
#include <arrayfire.h>
#include
int main(int argc, char const* argv[]) {
try { af::setBackend(AF_BACKEND_CUDA); af::setDevice(0); af::info(); float input_vals[25] = {0.0168, 0.0278, 0.0317, 0.0248, 0.0131, 0.0197, 0.0321, 0.0362, 0.0279, 0.0141, 0.0218, 0.0353, 0.0394, 0.0297, 0.0143, 0.0224, 0.0363, 0.0404, 0.0302, 0.0142, 0.0217, 0.0355, 0.0398, 0.0302, 0.0144}; af::array input(5, 5, input_vals); af::print("input", input); unsigned idx; float max_val; af::max<float>(&max_val, &idx, input); std::cout << "idx: " << idx << ", max_val: " << max_val << std::endl; //OK unsigned row = idx % input.dims(0); unsigned col = idx / input.dims(0); af::array input_cropped = input(af::seq(row - 1, row + 1), af::seq(col - 1, col + 1)); af::print("input_cropped", input_cropped); af::max<float>(&max_val, &idx, input_cropped); std::cout << "idx: " << idx << ", max_val: " << max_val << std::endl; //NOK, wrong idx (6 instead of 4), right maximum value af::print("input_cropped.copy()", input_cropped.copy()); af::max<float>(&max_val, &idx, input_cropped.copy()); std::cout << "idx: " << idx << ", max_val: " << max_val << std::endl; //OK } catch (const af::exception& e) { std::cerr << e.what() << '\n'; } return 0;}
`
.\test.exe
ArrayFire v3.9.0 (CUDA, 64-bit Windows, build b59a1ae)
Platform: CUDA Runtime 12.2, Driver: 12080
[0] NVIDIA RTX A2000 8GB Laptop GPU, 8192 MB, CUDA Compute 8.6
input
[5 5 1 1]
0.0168 0.0197 0.0218 0.0224 0.0217
0.0278 0.0321 0.0353 0.0363 0.0355
0.0317 0.0362 0.0394 0.0404 0.0398
0.0248 0.0279 0.0297 0.0302 0.0302
0.0131 0.0141 0.0143 0.0142 0.0144
idx: 17, max_val: 0.0404
input_cropped
[3 3 1 1]
0.0353 0.0363 0.0355
0.0394 0.0404 0.0398
0.0297 0.0302 0.0302
idx: 6, max_val: 0.0404
input_cropped.copy()
[3 3 1 1]
0.0353 0.0363 0.0355
0.0394 0.0404 0.0398
0.0297 0.0302 0.0302
idx: 4, max_val: 0.0404
System Information
3.9.0 on Windows 10
Checklist