FazBrowse GitHub Viewer
|
Trending
|
URL:
|
Home
Tools:
[Download Repo ZIP]
[View Raw Code]
[Original HTTPS Page]
CaffeOnACL/include/caffe/syncedmem.hpp at master · daeinki/CaffeOnACL · GitHub
daeinki
CaffeOnACL
Repository navigation
Code
Pull requests
Actions
Projects
Wiki
Security and quality
Insights
Expand file tree
Breadcrumbs
CaffeOnACL
/
include
/
caffe
/
syncedmem.hpp
Copy path
More file actions
More file actions
Latest commit
History
History
History
95 lines (81 loc) · 2.09 KB
Breadcrumbs
CaffeOnACL
/
include
/
caffe
/
syncedmem.hpp
Copy path
File metadata and controls
95 lines (81 loc) · 2.09 KB
Raw
Copy raw file
Download raw file
Open symbols panel
Edit and raw actions
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
#
ifndef
CAFFE_SYNCEDMEM_HPP_
#
define
CAFFE_SYNCEDMEM_HPP_
#
include
<
cstdlib
>
#
ifdef
USE_MKL
#
include
"
mkl.h
"
#
endif
#
include
"
caffe/common.hpp
"
namespace
caffe
{
//
If CUDA is available and in GPU mode, host memory will be allocated pinned,
//
using cudaMallocHost. It avoids dynamic pinning for transfers (DMA).
//
The improvement in performance seems negligible in the single GPU case,
//
but might be more significant for parallel training. Most importantly,
//
it improved stability for large models on many GPUs.
inline
void
CaffeMallocHost
(
void
** ptr,
size_t
size,
bool
* use_cuda) {
#
ifndef
CPU_ONLY
if
(
Caffe::mode
() == Caffe::
GPU
) {
CUDA_CHECK
(
cudaMallocHost
(ptr, size));
*use_cuda =
true
;
return
;
}
#
endif
#
ifdef
USE_MKL
*ptr =
mkl_malloc
(size ? size:
1
,
64
);
#
else
*ptr =
malloc
(size);
#
endif
*use_cuda =
false
;
CHECK
(*ptr) <<
"
host allocation of size
"
<< size <<
"
failed
"
;
}
inline
void
CaffeFreeHost
(
void
* ptr,
bool
use_cuda) {
#
ifndef
CPU_ONLY
if
(use_cuda) {
CUDA_CHECK
(
cudaFreeHost
(ptr));
return
;
}
#
endif
#
ifdef
USE_MKL
mkl_free
(ptr);
#
else
free
(ptr);
#
endif
}
/*
*
* @brief Manages memory allocation and synchronization between the host (CPU)
* and device (GPU).
*
* TODO(dox): more thorough description.
*/
class
SyncedMemory
{
public:
SyncedMemory
();
explicit
SyncedMemory
(
size_t
size);
~SyncedMemory
();
const
void
*
cpu_data
();
void
set_cpu_data
(
void
* data);
const
void
*
gpu_data
();
void
set_gpu_data
(
void
* data);
void
*
mutable_cpu_data
();
void
*
mutable_gpu_data
();
enum
SyncedHead {
UNINITIALIZED
,
HEAD_AT_CPU
,
HEAD_AT_GPU
,
SYNCED
};
SyncedHead
head
() {
return
head_; }
size_t
size
() {
return
size_; }
#
ifndef
CPU_ONLY
void
async_gpu_push
(
const
cudaStream_t& stream);
#
endif
private:
void
check_device
();
void
to_cpu
();
void
to_gpu
();
void
* cpu_ptr_;
void
* gpu_ptr_;
size_t
size_;
SyncedHead head_;
bool
own_cpu_data_;
bool
cpu_malloc_use_cuda_;
bool
own_gpu_data_;
int
device_;
DISABLE_COPY_AND_ASSIGN
(SyncedMemory);
};
//
class SyncedMemory
}
//
namespace caffe
#
endif
//
CAFFE_SYNCEDMEM_HPP_
Back
|
FazBrowse Home
|
New Git URL