FazBrowse GitHub Viewer | Trending |
URL:
| Home
Tools: [Download Repo ZIP]   [Original HTTPS Page]

Update CUDA driver version checks by umar456 · Pull Request #3161 · arrayfire/arrayfire · GitHub

Repository navigation

Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension .cpp  (5) .mk  (1) All 2 file types selected
Viewed files
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Unified
Split
Hide whitespace
Diff view
Unified
Split
Hide whitespace
7 changes: 0 additions & 7 deletions docs/doxygen.mk
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -1087,13 +1087,6 @@ VERBATIM_HEADERS = YES

ALPHABETICAL_INDEX = YES

# The COLS_IN_ALPHA_INDEX tag can be used to specify the number of columns in
# which the alphabetical index list will be split.
# Minimum value: 1, maximum value: 20, default value: 5.
# This tag requires that the tag ALPHABETICAL_INDEX is set to YES.

COLS_IN_ALPHA_INDEX = 5

# In case all classes in a project start with a common prefix, all classes will
# be put under the same header in the alphabetical index. The IGNORE_PREFIX tag
# can be used to specify a prefix (or a list of prefixes) that should be ignored
Expand Down
11 changes: 11 additions & 0 deletions src/backend/common/DefaultMemoryManager.cpp
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,8 @@
#include <af/event.h>
#include <af/memory.h>

#include <algorithm>
#include <cstdio>
#include <memory>
#include <string>
#include <vector>
Expand Down Expand Up @@ -121,6 +123,8 @@ void DefaultMemoryManager::setMaxMemorySize() {
memsize == 0
? ONE_GB
: max(memsize * 0.75, static_cast<double>(memsize - ONE_GB));
AF_TRACE("memory[{}].max_bytes: {}", n,
bytesToString(memory[n].max_bytes));
}
}

Expand Down Expand Up @@ -161,6 +165,13 @@ void *DefaultMemoryManager::alloc(bool user_lock, const unsigned ndims,
// Perhaps look at total memory available as a metric
if (current.lock_bytes >= current.max_bytes ||
current.total_buffers >= this->max_buffers) {
AF_TRACE(
"Running GC: current.lock_bytes({}) >= "
"current.max_bytes({}) || current.total_buffers({}) >= "
"this->max_buffers({})\n",
current.lock_bytes, current.max_bytes,
current.total_buffers, this->max_buffers);

this->signalMemoryCleanup();
}

Expand Down
11 changes: 6 additions & 5 deletions src/backend/cpu/platform.cpp
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -15,10 +15,11 @@
#include <version.hpp>
#include <af/version.h>

#include <algorithm>
#include <cctype>
#include <cstdio>
#include <memory>
#include <sstream>
#include <string>

using common::memory::MemoryManagerBase;
using std::endl;
Expand Down Expand Up @@ -110,7 +111,7 @@ int& getMaxJitSize() {
if (length <= 0) {
string env_var = getEnvVar("AF_CPU_MAX_JIT_LEN");
if (!env_var.empty()) {
int input_len = std::stoi(env_var);
int input_len = stoi(env_var);
length = input_len > 0 ? input_len : MAX_JIT_LEN;
} else {
length = MAX_JIT_LEN;
Expand Down Expand Up @@ -161,15 +162,15 @@ MemoryManagerBase& memoryManager() {
}

void setMemoryManager(unique_ptr<MemoryManagerBase> mgr) {
return DeviceManager::getInstance().setMemoryManager(std::move(mgr));
return DeviceManager::getInstance().setMemoryManager(move(mgr));
}

void resetMemoryManager() {
return DeviceManager::getInstance().resetMemoryManager();
}

void setMemoryManagerPinned(std::unique_ptr<MemoryManagerBase> mgr) {
return DeviceManager::getInstance().setMemoryManagerPinned(std::move(mgr));
void setMemoryManagerPinned(unique_ptr<MemoryManagerBase> mgr) {
return DeviceManager::getInstance().setMemoryManagerPinned(move(mgr));
}

void resetMemoryManagerPinned() {
Expand Down
75 changes: 41 additions & 34 deletions src/backend/cuda/device_manager.cpp
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -38,15 +38,13 @@

#include <algorithm>
#include <array>
#include <cstdio>
#include <memory>
#include <mutex>
#include <sstream>
#include <stdexcept>
#include <string>
#include <thread>
#include <utility>
#include <vector>

using std::begin;
using std::end;
Expand Down Expand Up @@ -97,6 +95,7 @@ static const int jetsonComputeCapabilities[] = {

// clang-format off
static const cuNVRTCcompute Toolkit2MaxCompute[] = {
{11040, 8, 6, 0},
{11030, 8, 6, 0},
{11020, 8, 6, 0},
{11010, 8, 6, 0},
Expand All @@ -112,12 +111,23 @@ static const cuNVRTCcompute Toolkit2MaxCompute[] = {
{ 7000, 5, 2, 3}};
// clang-format on

// A tuple of Compute Capability and the associated number of cores in each
// streaming multiprocessors for that architecture
struct ComputeCapabilityToStreamingProcessors {
// The compute capability in hex
// 0xMm (hex), M = major version, m = minor version
int compute_capability;
// Number of CUDA cores per SM
int cores_per_sm;
};

/// Map giving the minimum device driver needed in order to run a given version
/// of CUDA for both Linux/Mac and Windows from:
/// https://docs.nvidia.com/cuda/cuda-toolkit-release-notes/index.html
// clang-format off
static const ToolkitDriverVersions
CudaToDriverVersion[] = {
{11040, 470.42f, 471.11f},
{11030, 465.19f, 465.89f},
{11020, 460.27f, 460.82f},
{11010, 455.23f, 456.38f},
Expand All @@ -133,6 +143,35 @@ static const ToolkitDriverVersions
{7000, 346.46f, 347.62f}};
// clang-format on

// Vector of minimum supported compute versions for CUDA toolkit (i+1).*
// where i is the index of the vector
static const std::array<int, 11> minSV{{1, 1, 1, 1, 1, 1, 2, 2, 3, 3, 3}};

static ComputeCapabilityToStreamingProcessors gpus[] = {
{0x10, 8}, {0x11, 8}, {0x12, 8}, {0x13, 8}, {0x20, 32},
{0x21, 48}, {0x30, 192}, {0x32, 192}, {0x35, 192}, {0x37, 192},
{0x50, 128}, {0x52, 128}, {0x53, 128}, {0x60, 64}, {0x61, 128},
{0x62, 128}, {0x70, 64}, {0x75, 64}, {0x80, 64}, {0x86, 128},
{-1, -1},
};

// pulled from CUTIL from CUDA SDK
static inline int compute2cores(unsigned major, unsigned minor) {

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Choose a reason Spam Abuse Off Topic Outdated Duplicate Resolved Low Quality

Please move this function movement changes into cleanup commit.

for (int i = 0; gpus[i].compute_capability != -1; ++i) {
if (static_cast<unsigned>(gpus[i].compute_capability) ==
(major << 4U) + minor) {
return gpus[i].cores_per_sm;
}
}
return 0;
}

static inline int getMinSupportedCompute(int cudaMajorVer) {
int CVSize = static_cast<int>(minSV.size());
return (cudaMajorVer > CVSize ? minSV[CVSize - 1]
: minSV[cudaMajorVer - 1]);
}

bool isEmbedded(pair<int, int> compute) {
int version = compute.first * 1000 + compute.second * 10;
return end(jetsonComputeCapabilities) !=
Expand Down Expand Up @@ -234,27 +273,6 @@ pair<int, int> getComputeCapability(const int device) {
return DeviceManager::getInstance().devJitComputes[device];
}

// pulled from CUTIL from CUDA SDK
static inline int compute2cores(unsigned major, unsigned minor) {
struct {
int compute; // 0xMm (hex), M = major version, m = minor version
int cores;
} gpus[] = {
{0x10, 8}, {0x11, 8}, {0x12, 8}, {0x13, 8}, {0x20, 32},
{0x21, 48}, {0x30, 192}, {0x32, 192}, {0x35, 192}, {0x37, 192},
{0x50, 128}, {0x52, 128}, {0x53, 128}, {0x60, 64}, {0x61, 128},
{0x62, 128}, {0x70, 64}, {0x75, 64}, {0x80, 64}, {0x86, 128},
{-1, -1},
};

for (int i = 0; gpus[i].compute != -1; ++i) {
if (static_cast<unsigned>(gpus[i].compute) == (major << 4U) + minor) {
return gpus[i].cores;
}
}
return 0;
}

// Return true if greater, false if lesser.
// if equal, it continues to next comparison
#define COMPARE(a, b, f) \
Expand Down Expand Up @@ -312,17 +330,6 @@ static inline bool card_compare_num(const cudaDevice_t &l,
return false;
}

static inline int getMinSupportedCompute(int cudaMajorVer) {
// Vector of minimum supported compute versions
// for CUDA toolkit (i+1).* where i is the index
// of the vector
static const std::array<int, 10> minSV{{1, 1, 1, 1, 1, 1, 2, 2, 3, 3}};

int CVSize = static_cast<int>(minSV.size());
return (cudaMajorVer > CVSize ? minSV[CVSize - 1]
: minSV[cudaMajorVer - 1]);
}

bool DeviceManager::checkGraphicsInteropCapability() {
static std::once_flag checkInteropFlag;
thread_local bool capable = true;
Expand Down
Loading

Back | FazBrowse Home | New Git URL