Skip to content

Commit 426c093

Browse files
o-aduh95
authored andcommitted
deps: V8: cherry-pick 7f80b64d1443
Original commit message: [base] Optimize Relaxed_Memset with word-sized atomic stores When filling memory with relaxed atomic stores for types smaller than a word (e.g., Tagged_t) clang does not merge the stores. This change optimizes Relaxed_Memset to store word-sized chunks after an alignment prologue and before a trailing epilogue. __attribute__((may_alias)) is used since otherwise aliasing pointers of different size are UB in C++. On x64 release builds, this achieves ~1.7x speedup for Array.prototype.fill with Smi value. Bug: 540352782 TAG=agy CONV=d5bd36e5-0b77-41b5-8bfb-68b320ea0546 Change-Id: I3be81d687d30a13c67ffed6cd4a6b4366411a6f6 Reviewed-on: https://chromium-review.googlesource.com/c/v8/v8/+/8177163 Commit-Queue: Nico Hartmann <nicohartmann@chromium.org> Reviewed-by: Nico Hartmann <nicohartmann@chromium.org> Auto-Submit: Olivier Flückiger <olivf@chromium.org> Cr-Commit-Position: refs/heads/main@{#109025} Refs: v8/v8@7f80b64
1 parent 74c1ca1 commit 426c093

3 files changed

Lines changed: 120 additions & 3 deletions

File tree

‎common.gypi‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -44,7 +44,7 @@
4444

4545
# Reset this number to 0 on major V8 upgrades.
4646
# Increment by one for each non-official patch applied to deps/v8.
47-
'v8_embedder_string': '-node.22',
47+
'v8_embedder_string': '-node.23',
4848

4949
##### V8 defaults for Node.js #####
5050

‎deps/v8/src/base/memcopy.h‎

Lines changed: 51 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,8 @@
99

1010
#include <algorithm>
1111
#include <atomic>
12+
#include <limits>
13+
#include <type_traits>
1214

1315
#include "include/v8config.h"
1416
#include "src/base/base-export.h"
@@ -215,11 +217,59 @@ inline void MemCopyAndSwitchEndianness(void* dst, const void* src,
215217
}
216218
#endif
217219

220+
// Helper to replicate an integral `value` across all bytes of a uintptr_t word.
221+
template <typename T>
222+
requires std::is_integral_v<T>
223+
constexpr uintptr_t ReplicateValueWord(T value) {
224+
if constexpr (sizeof(T) >= sizeof(uintptr_t)) {
225+
return static_cast<uintptr_t>(value);
226+
} else {
227+
using UnsignedT = std::conditional_t<
228+
sizeof(T) == 1, uint8_t,
229+
std::conditional_t<
230+
sizeof(T) == 2, uint16_t,
231+
std::conditional_t<sizeof(T) == 4, uint32_t, uint64_t>>>;
232+
constexpr uintptr_t kMultiplier =
233+
static_cast<uintptr_t>(-1) / std::numeric_limits<UnsignedT>::max();
234+
return static_cast<uintptr_t>(static_cast<UnsignedT>(value)) * kMultiplier;
235+
}
236+
}
237+
218238
// Fills `destination` with `count` `value`s.
219239
template <typename T>
220-
inline void Relaxed_Memset(T* destination, T value, size_t count)
240+
V8_INLINE void Relaxed_Memset(T* destination, T value, size_t count)
221241
requires std::is_integral_v<T>
222242
{
243+
#if defined(__ATOMIC_RELAXED) && (defined(__GNUC__) || defined(__clang__))
244+
if constexpr (sizeof(T) < sizeof(uintptr_t)) {
245+
constexpr size_t kElementsPerWord = sizeof(uintptr_t) / sizeof(T);
246+
// Prologue: store elements until `destination` is word-aligned.
247+
while (reinterpret_cast<uintptr_t>(destination) % sizeof(uintptr_t) != 0 &&
248+
count > 0) {
249+
std::atomic_ref<T>(*destination).store(value, std::memory_order_relaxed);
250+
destination++;
251+
count--;
252+
}
253+
// Main loop: store word-sized chunks using may_alias attribute and
254+
// __atomic_store_n.
255+
using AliasedWord = uintptr_t __attribute__((may_alias));
256+
const uintptr_t word_val = ReplicateValueWord<T>(value);
257+
AliasedWord* word_dest = reinterpret_cast<AliasedWord*>(destination);
258+
size_t words = count / kElementsPerWord;
259+
for (size_t w = 0; w < words; w++) {
260+
__atomic_store_n(word_dest + w, word_val, __ATOMIC_RELAXED);
261+
}
262+
destination += words * kElementsPerWord;
263+
count -= words * kElementsPerWord;
264+
// Epilogue: store remaining trailing elements.
265+
while (count > 0) {
266+
std::atomic_ref<T>(*destination).store(value, std::memory_order_relaxed);
267+
destination++;
268+
count--;
269+
}
270+
return;
271+
}
272+
#endif
223273
for (size_t i = 0; i < count; i++) {
224274
std::atomic_ref<T>(destination[i]).store(value, std::memory_order_relaxed);
225275
}

‎deps/v8/test/unittests/base/atomic-utils-unittest.cc‎

Lines changed: 68 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -2,9 +2,11 @@
22
// Use of this source code is governed by a BSD-style license that can be
33
// found in the LICENSE file.
44

5+
#include "src/base/atomic-utils.h"
6+
57
#include <limits.h>
68

7-
#include "src/base/atomic-utils.h"
9+
#include "src/base/memcopy.h"
810
#include "src/base/platform/platform.h"
911
#include "testing/gtest/include/gtest/gtest.h"
1012

@@ -236,5 +238,70 @@ TEST(AsAtomicWord, SetBits_Concurrent) {
236238
}
237239
}
238240

241+
template <typename T>
242+
void TestRelaxedMemsetHelper(T value, T sentinel) {
243+
constexpr int kBufferSize = 64;
244+
for (int offset = 0; offset < 8; ++offset) {
245+
for (int count = 0; count < 32; ++count) {
246+
if (offset + count > kBufferSize) continue;
247+
alignas(8) T buffer[kBufferSize];
248+
for (int i = 0; i < kBufferSize; ++i) {
249+
buffer[i] = sentinel;
250+
}
251+
Relaxed_Memset(&buffer[offset], value, count);
252+
for (int i = 0; i < kBufferSize; ++i) {
253+
if (i >= offset && i < offset + count) {
254+
EXPECT_EQ(value, buffer[i])
255+
<< "offset=" << offset << " count=" << count << " i=" << i;
256+
} else {
257+
EXPECT_EQ(sentinel, buffer[i])
258+
<< "offset=" << offset << " count=" << count << " i=" << i;
259+
}
260+
}
261+
}
262+
}
263+
}
264+
265+
TEST(RelaxedMemset, VariousTypesAndAlignments) {
266+
TestRelaxedMemsetHelper<uint8_t>(0xAB, 0x12);
267+
TestRelaxedMemsetHelper<uint16_t>(0xABCD, 0x1234);
268+
TestRelaxedMemsetHelper<uint32_t>(0xABCDEF01u, 0x12345678u);
269+
TestRelaxedMemsetHelper<uint64_t>(0xABCDEF0123456789ULL,
270+
0x1111222233334444ULL);
271+
TestRelaxedMemsetHelper<int8_t>(-1, 0);
272+
TestRelaxedMemsetHelper<int32_t>(-123456, 789);
273+
}
274+
275+
TEST(RelaxedMemset, ConstexprReplicatedWord) {
276+
static_assert(ReplicateValueWord<uint8_t>(0xAB) ==
277+
#if V8_HOST_ARCH_64_BIT
278+
0xABABABABABABABABULL
279+
#else
280+
0xABABABABu
281+
#endif
282+
);
283+
static_assert(ReplicateValueWord<uint16_t>(0x1234) ==
284+
#if V8_HOST_ARCH_64_BIT
285+
0x1234123412341234ULL
286+
#else
287+
0x12341234u
288+
#endif
289+
);
290+
291+
constexpr int kBufferSize = 32;
292+
alignas(8) uint8_t buffer[kBufferSize];
293+
for (int i = 0; i < kBufferSize; ++i) {
294+
buffer[i] = 0xFF;
295+
}
296+
Relaxed_Memset<uint8_t>(&buffer[1], 0x42, 15);
297+
EXPECT_EQ(0xFF, buffer[0]);
298+
for (int i = 1; i < 16; ++i) {
299+
EXPECT_EQ(0x42, buffer[i]);
300+
}
301+
for (int i = 16; i < kBufferSize; ++i) {
302+
EXPECT_EQ(0xFF, buffer[i]);
303+
}
304+
}
305+
239306
} // namespace base
240307
} // namespace v8

0 commit comments

Comments
 (0)