The execution model of the GPU expects that groups of threads will execute in lock-step in SIMD fashion. It's both important for performance and correctness that we treat this as the smallest possible granularity for an RPC operation. Thus, we map multiple threads to a single larger buffer and ship that across the wire. This patch makes the necessary changes to support executing the RPC on the GPU with multiple threads. This requires some workarounds to mimic the model when handling the protocol from the CPU. I'm not completely happy with some of the workarounds required, but I think it should work. Uses some of the implementation details from D148191. Reviewed By: JonChesterfield Differential Revision: https://reviews.llvm.org/D148943
50 lines
1.5 KiB
C++
50 lines
1.5 KiB
C++
//===-- Loader test to check the RPC interface with the loader ------------===//
|
|
//
|
|
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
|
// See https://llvm.org/LICENSE.txt for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
//
|
|
//===----------------------------------------------------------------------===//
|
|
|
|
#include "src/__support/GPU/utils.h"
|
|
#include "src/__support/RPC/rpc_client.h"
|
|
#include "test/IntegrationTest/test.h"
|
|
|
|
using namespace __llvm_libc;
|
|
|
|
static void test_add_simple() {
|
|
uint32_t num_additions =
|
|
10 + 10 * gpu::get_thread_id() + 10 * gpu::get_block_id();
|
|
uint64_t cnt = 0;
|
|
for (uint32_t i = 0; i < num_additions; ++i) {
|
|
rpc::Client::Port port = rpc::client.open(rpc::TEST_INCREMENT);
|
|
port.send_and_recv(
|
|
[=](rpc::Buffer *buffer) {
|
|
reinterpret_cast<uint64_t *>(buffer->data)[0] = cnt;
|
|
},
|
|
[&](rpc::Buffer *buffer) {
|
|
cnt = reinterpret_cast<uint64_t *>(buffer->data)[0];
|
|
});
|
|
port.close();
|
|
}
|
|
ASSERT_TRUE(cnt == num_additions && "Incorrect sum");
|
|
}
|
|
|
|
// Test to ensure that the RPC mechanism doesn't hang on divergence.
|
|
static void test_noop(uint8_t data) {
|
|
rpc::Client::Port port = rpc::client.open(rpc::NOOP);
|
|
port.send([=](rpc::Buffer *buffer) { buffer->data[0] = data; });
|
|
port.close();
|
|
}
|
|
|
|
TEST_MAIN(int argc, char **argv, char **envp) {
|
|
test_add_simple();
|
|
|
|
if (gpu::get_thread_id() % 2)
|
|
test_noop(1);
|
|
else
|
|
test_noop(2);
|
|
|
|
return 0;
|
|
}
|