1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
|
#include <new>
#include <array>
#include <atomic>
#include <cstddef>
#include <benchmark/benchmark.h>
enum class Alignment: std::size_t {
Default = alignof(std::atomic<std::size_t>),
Cacheline = std::hardware_destructive_interference_size
};
template <typename T,
Alignment Align = Alignment::Default,
std::size_t Size = 1024>
class RingBuffer {
public:
__attribute__((noinline)) bool Push(const T& value) {
const std::size_t writer = writer_.load(std::memory_order_relaxed);
const std::size_t next_writer = (writer + 1) % Size;
if (next_writer == reader_.load(std::memory_order_acquire))
return false;
buffer_[writer] = value;
writer_.store(next_writer, std::memory_order_release);
return true;
}
__attribute__((noinline)) bool Pop(T& value) {
const std::size_t reader = reader_.load(std::memory_order_relaxed);
const std::size_t next_reader = (reader + 1) % Size;
if (reader == writer_.load(std::memory_order_acquire))
return false;
value = buffer_[reader];
reader_.store(next_reader, std::memory_order_release);
return true;
}
private:
alignas(Align) std::atomic<std::size_t> writer_{0};
alignas(Align) std::atomic<std::size_t> reader_{0};
alignas(Align) std::array<T, Size> buffer_;
};
// See: https://rigtorp.se/ringbuffer
template <typename T,
Alignment Align = Alignment::Default,
std::size_t Size = 1024>
class CachedRingBuffer {
public:
__attribute__((noinline)) bool Push(const T& value) {
const std::size_t writer = writer_.load(std::memory_order_relaxed);
const std::size_t next_writer = (writer + 1) % Size;
if (next_writer == cached_reader_) {
cached_reader_ = reader_.load(std::memory_order_acquire);
if (next_writer == cached_reader_)
return false;
}
buffer_[writer] = value;
writer_.store(next_writer, std::memory_order_release);
return true;
}
__attribute__((noinline)) bool Pop(T& value) {
const std::size_t reader = reader_.load(std::memory_order_relaxed);
const std::size_t next_reader = (reader + 1) % Size;
if (reader == cached_writer_) {
cached_writer_ = writer_.load(std::memory_order_acquire);
if (reader == cached_writer_)
return false;
}
value = buffer_[reader];
reader_.store(next_reader, std::memory_order_release);
return true;
}
private:
alignas(Align) std::atomic<std::size_t> writer_{0};
alignas(Align) std::size_t cached_reader_{0};
alignas(Align) std::atomic<std::size_t> reader_{0};
alignas(Align) std::size_t cached_writer_{0};
alignas(Align) std::array<T, Size> buffer_;
};
static bool IsPusherThread(benchmark::State& state) {
return state.thread_index() == 0;
}
template <typename RingBuffer>
static void DoPushPop(benchmark::State& state) {
// Make ring_buffer static to share it between threads.
static RingBuffer ring_buffer;
if (IsPusherThread(state)) {
int x = 0;
for (auto _ : state) {
while (!ring_buffer.Push(x))
++x;
}
} else {
for (int x = 0; auto _ : state) {
while (!ring_buffer.Pop(x)) {}
benchmark::DoNotOptimize(x);
}
}
}
static void BM_PushPopNoAlign(benchmark::State& state) {
DoPushPop<RingBuffer<int>>(state);
}
BENCHMARK(BM_PushPopNoAlign)->Threads(2);
static void BM_PushPopCachelineAlign(benchmark::State& state) {
DoPushPop<RingBuffer<int, Alignment::Cacheline>>(state);
}
BENCHMARK(BM_PushPopCachelineAlign)->Threads(2);
static void BM_PushPopCachedNoAlign(benchmark::State& state) {
DoPushPop<CachedRingBuffer<int>>(state);
}
BENCHMARK(BM_PushPopCachedNoAlign)->Threads(2);
static void BM_PushPopCachedCachelineAlign(benchmark::State& state) {
DoPushPop<CachedRingBuffer<int, Alignment::Cacheline>>(state);
}
BENCHMARK(BM_PushPopCachedCachelineAlign)->Threads(2);
BENCHMARK_MAIN();
|