Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions .github/workflows/build-tenstorrent.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
name: Build and Test Tenstorrent Backend

on:
pull_request:
branches: [main, spatter-devel]
schedule:
- cron: '30 8 * * *'

jobs:
build-tenstorrent:
runs-on: self-hosted
steps:
- uses: actions/checkout@v4
- name: Run batch file
run: cd tests/misc && chmod +x run-crnch-tenstorrent.sh && sbatch run-crnch-tenstorrent.sh
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -9,3 +9,4 @@ src/python/env
*.pyc
spatter-cuda-test.out
json.tar.xz
generated/
3 changes: 3 additions & 0 deletions AUTHORS
Original file line number Diff line number Diff line change
Expand Up @@ -20,3 +20,6 @@ Vincent Huang

James Wood
OneAPI backend support

Roy Cheng
Tenstorrent backend
52 changes: 51 additions & 1 deletion Build.md
Original file line number Diff line number Diff line change
Expand Up @@ -28,4 +28,54 @@
* avx_crossplatform
* non_avx
* `-DUSE_MPI=1`
* `-DUSE_PAPI=1`
* `-DUSE_PAPI=1`

## Tenstorrent
Blackhole and other Tenstorrent accelerators, via tt-metal. Enable with
`-DUSE_TENSTORRENT=ON`.

Device kernels are compiled at run time by tt-metal's vendored sfpi toolchain,
so no extra device compiler is needed at build time. The host side needs a C++20
compiler and `libtt_metal`.

### Optional arguments
* `-DTT_METAL_LIB_DIR=<DIR>` - directory holding `libtt_metal.so`
* `-DTT_METAL_INCLUDE_DIRS=<DIR;DIR;...>` - include roots for the host headers

### Runtime environment
* `TT_METAL_RUNTIME_ROOT` must point at the directory containing `tt_metal/`,
otherwise tt-metal aborts with "Root Directory is not set" before it opens a
device.
* `TT_VISIBLE_DEVICES` selects which chip(s) to use on a multi-card host.

### If you installed Tenstorrent support with a pip `ttnn` wheel
The wheel ships `libtt_metal.so` and the device-kernel headers, but **not** the
host API headers, so CMake will find the library and then report the headers
missing. They can be assembled without root and without building tt-metal:

```
git clone --depth 1 --branch <version> --filter=blob:none --sparse \
https://github.com/tenstorrent/tt-metal ~/ttmetal-src
cd ~/ttmetal-src && git sparse-checkout set tt_metal tt_stl
git submodule update --init --depth 1 tt_metal/third_party/umd
```

That covers `tt-metalium`, `tt_stl`, `hostdevcommon` and `umd`. The remaining
four are header-only and are fetched by tt-metal's build rather than vendored,
so clone them at the versions pinned in `~/ttmetal-src/third_party/CMakeLists.txt`:
fmt, nlohmann/json, tt-logger and spdlog, plus enchantum. Pass every include
root via `-DTT_METAL_INCLUDE_DIRS`.

Do **not** run tt-metal's own CMake configure just to obtain these: it pulls
system packages (boost, capnproto, protobuf) that require root, whereas the
header-only clones do not.

### Notes
* Blackhole has no FP64 fabric. This does not affect Spatter, because gather and
scatter perform no arithmetic; each element is moved as an opaque 8-byte
payload.
* The DRAM allocator aligns pages to 64 bytes, so a buffer with one 8-byte
element per page occupies 8x its logical size on device. Size `-l` accordingly.
* `gather` and `scatter` are implemented. The `multi_*` and atomic variants are
not yet, and report so at run time.

1 change: 1 addition & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@ include(pkgs/JSONSupport)
include(pkgs/MPISupport)
include(pkgs/OpenMPSupport)
include(pkgs/CUDASupport)
include(pkgs/TenstorrentSupport)

if (APPLE)
set(CMAKE_INSTALL_RPATH "@executable_path/../lib")
Expand Down
8 changes: 7 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,7 @@ CMake is required to build Spatter. Currently we require CMake 3.25 or newer.

To build with CMake from the main source directory, use the following command structure:
```
cmake -DCMAKE_BUILD_TYPE=<BUILD_TYPE> -DUSE_<OPENMP/CUDA/MPI>=1 -B build_<BACKEND> -S .
cmake -DCMAKE_BUILD_TYPE=<BUILD_TYPE> -DUSE_<OPENMP/CUDA/MPI/TENSTORRENT>=1 -B build_<BACKEND> -S .
cd build_<BACKEND>
make
```
Expand All @@ -63,6 +63,12 @@ For CUDA builds, we normally load CUDA 11/12 using NVHPC:
```
cmake -DUSE_CUDA=1 -B build_cuda -S .
```

For Tenstorrent builds (see `Build.md` for the header requirements):

```
cmake -DUSE_TENSTORRENT=ON -B build_tenstorrent -S .
```
For a complete list of build options, see [Build.md](Build.md)

## Running Spatter
Expand Down
79 changes: 79 additions & 0 deletions cmake/pkgs/TenstorrentSupport.cmake
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
# Tenstorrent device kernels are compiled at run time by tt-metal's vendored
# sfpi toolchain, so there is no device compiler to enable here; only the host
# library matters at build time. There is no upstream FindTTMetal module.

option(USE_TENSTORRENT "Enable support for Tenstorrent accelerators")

if (USE_TENSTORRENT)
set(TT_METAL_INCLUDE_DIRS "" CACHE STRING
"Semicolon-separated include roots for tt-metal host headers")
set(TT_METAL_LIB_DIR "" CACHE PATH "Directory containing libtt_metal.so")

find_package(Python3 COMPONENTS Interpreter QUIET)
if (Python3_Interpreter_FOUND)
execute_process(
COMMAND ${Python3_EXECUTABLE} -c
"import ttnn, os; print(os.path.dirname(ttnn.__file__))"
OUTPUT_VARIABLE TTNN_DIR
OUTPUT_STRIP_TRAILING_WHITESPACE
ERROR_QUIET)
if (TTNN_DIR)
message(STATUS "Tenstorrent: found ttnn wheel at ${TTNN_DIR}")
endif()
endif()

find_library(TT_METAL_LIB
NAMES tt_metal
HINTS ${TT_METAL_LIB_DIR}
${TTNN_DIR}/build/lib
$ENV{TT_METAL_HOME}/build/lib
$ENV{TT_METAL_RUNTIME_ROOT}/build/lib)

# Headers include each other as <tt-metalium/...>, so the root is the parent
# of tt-metalium/.
find_path(TT_METALIUM_INCLUDE_DIR
NAMES tt-metalium/host_api.hpp
HINTS ${TT_METAL_INCLUDE_DIRS}
$ENV{TT_METAL_HOME}/tt_metal/api
${TTNN_DIR}/tt_metal/api)

if (TT_METAL_LIB AND TT_METALIUM_INCLUDE_DIR)
message(STATUS "Found libtt_metal: ${TT_METAL_LIB}")
message(STATUS "Found tt-metalium headers: ${TT_METALIUM_INCLUDE_DIR}")

set(CMAKE_CXX_STANDARD 20)
set(CMAKE_CXX_STANDARD_REQUIRED ON)

include_directories(${TT_METALIUM_INCLUDE_DIR})
if (TT_METAL_INCLUDE_DIRS)
include_directories(${TT_METAL_INCLUDE_DIRS})
endif()

set(COMMON_LINK_LIBRARIES ${COMMON_LINK_LIBRARIES} ${TT_METAL_LIB})
add_definitions(-DUSE_TENSTORRENT)
else()
if (TT_METAL_LIB AND NOT TT_METALIUM_INCLUDE_DIR)
message(STATUS
"Tenstorrent: libtt_metal was found but the host headers were not. "
"A pip `ttnn` wheel ships the library and the device-kernel headers "
"but no host API headers. Point -DTT_METAL_INCLUDE_DIRS at a matching "
"tt-metal checkout plus its header-only dependencies, e.g.\n"
" git clone --depth 1 --branch v0.72.0 --filter=blob:none --sparse \\\n"
" https://github.com/tenstorrent/tt-metal ~/ttmetal-src\n"
" cd ~/ttmetal-src && git sparse-checkout set tt_metal tt_stl\n"
" git submodule update --init --depth 1 tt_metal/third_party/umd\n"
"then clone fmt, nlohmann/json, tt-logger, spdlog and enchantum at the "
"versions pinned in that checkout's third_party/CMakeLists.txt and pass "
"every include root. Do NOT run tt-metal's own configure for this: it "
"pulls system packages (boost, capnproto, protobuf) that need root, "
"whereas the header-only clones do not.")
elseif (NOT TT_METAL_LIB)
message(STATUS
"Tenstorrent: libtt_metal not found. Install tt-metal, or pip install "
"ttnn, or set -DTT_METAL_LIB_DIR.")
endif()
message(FATAL_ERROR
"USE_TENSTORRENT=ON but no usable Tenstorrent installation was found. "
"See the diagnostic above, or configure without -DUSE_TENSTORRENT=ON.")
endif()
endif()
21 changes: 21 additions & 0 deletions src/Spatter/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,19 @@ if (USE_CUDA)
set(CUDA_INCLUDE_FILES CudaBackend.hh)
endif()

if (USE_TENSTORRENT)
add_library(tenstorrent_backend SHARED TenstorrentBackend.cc)
target_link_libraries(tenstorrent_backend PUBLIC ${TT_METAL_LIB})
target_compile_features(tenstorrent_backend PUBLIC cxx_std_20)
# tt-metal's headers pull in fmt and spdlog; use them header-only so the
# backend does not need those libraries at link time.
target_compile_definitions(tenstorrent_backend PRIVATE FMT_HEADER_ONLY)
set(TENSTORRENT_INCLUDE_FILES TenstorrentBackend.hh)
endif()

set(SPATTER_INCLUDE_FILES
${CUDA_INCLUDE_FILES}
${TENSTORRENT_INCLUDE_FILES}
Configuration.hh
Input.hh
JSONParser.hh
Expand Down Expand Up @@ -61,6 +72,10 @@ if (USE_CUDA)
set(COMMON_LINK_LIBRARIES ${COMMON_LINK_LIBRARIES} cuda_backend)
endif()

if (USE_TENSTORRENT)
set(COMMON_LINK_LIBRARIES ${COMMON_LINK_LIBRARIES} tenstorrent_backend)
endif()

target_link_libraries(Spatter
PUBLIC
${COMMON_LINK_LIBRARIES}
Expand Down Expand Up @@ -93,6 +108,12 @@ if (USE_CUDA)
ARCHIVE DESTINATION lib)
endif()

if (USE_TENSTORRENT)
install (TARGETS tenstorrent_backend
LIBRARY DESTINATION lib
ARCHIVE DESTINATION lib)
endif()

# Library/Header installation section

#set(ConfigPackageLocation lib/cmake/Spatter)
Expand Down
146 changes: 146 additions & 0 deletions src/Spatter/Configuration.cc
Original file line number Diff line number Diff line change
Expand Up @@ -921,4 +921,150 @@ void Configuration<Spatter::CUDA>::setup() {
}
#endif

#ifdef USE_TENSTORRENT
Configuration<Spatter::Tenstorrent>::Configuration(const size_t id,
const std::string name, const std::string kernel,
const aligned_vector<size_t> &pattern,
const aligned_vector<size_t> &pattern_gather,
const aligned_vector<size_t> &pattern_scatter,
aligned_vector<double> &sparse, double *&dev_sparse, size_t &sparse_size,
aligned_vector<double> &sparse_gather, double *&dev_sparse_gather,
size_t &sparse_gather_size, aligned_vector<double> &sparse_scatter,
double *&dev_sparse_scatter, size_t &sparse_scatter_size,
aligned_vector<double> &dense,
aligned_vector<aligned_vector<double>> &dense_perthread, double *&dev_dense,
size_t &dense_size, const size_t delta, const size_t delta_gather,
const size_t delta_scatter, const long int seed, const size_t wrap,
const size_t count, const size_t shared_mem, const size_t local_work_size,
const unsigned long nruns, const bool aggregate, const bool atomic,
const unsigned long verbosity)
: ConfigurationBase(id, name, kernel, pattern, pattern_gather,
pattern_scatter, sparse, dev_sparse, sparse_size, sparse_gather,
dev_sparse_gather, sparse_gather_size, sparse_scatter,
dev_sparse_scatter, sparse_scatter_size, dense, dense_perthread,
dev_dense, dense_size, delta, delta_gather, delta_scatter, seed,
wrap, count, shared_mem, local_work_size, 1, nruns, aggregate, atomic,
false, false, verbosity),
dev_pattern(nullptr), dev_pattern_gather(nullptr),
dev_pattern_scatter(nullptr) {

setup();
}

Configuration<Spatter::Tenstorrent>::~Configuration() {
tt_device_free(dev_pattern);
tt_device_free(dev_pattern_gather);
tt_device_free(dev_pattern_scatter);

if (dev_sparse) {
tt_device_free(dev_sparse);
dev_sparse = nullptr;
}
if (dev_sparse_gather) {
tt_device_free(dev_sparse_gather);
dev_sparse_gather = nullptr;
}
if (dev_sparse_scatter) {
tt_device_free(dev_sparse_scatter);
dev_sparse_scatter = nullptr;
}
if (dev_dense) {
tt_device_free(dev_dense);
dev_dense = nullptr;
}
}

int Configuration<Spatter::Tenstorrent>::run(bool timed, unsigned long run_id) {
return ConfigurationBase::run(timed, run_id);
}

void Configuration<Spatter::Tenstorrent>::gather(
bool timed, unsigned long run_id) {
size_t pattern_length = pattern.size();

#ifdef USE_MPI
MPI_Barrier(MPI_COMM_WORLD);
#endif

// The wrapper is synchronous: it ends with Finish() on the mesh command
// queue, so no separate device synchronize is needed here.
float time_ms = tt_gather_wrapper(
dev_pattern, dev_sparse, dev_dense, pattern_length, delta, wrap, count);

if (timed)
time_seconds[run_id] = ((double)time_ms / 1000.0);
}

void Configuration<Spatter::Tenstorrent>::scatter(
bool timed, unsigned long run_id) {
size_t pattern_length = pattern.size();

#ifdef USE_MPI
MPI_Barrier(MPI_COMM_WORLD);
#endif

float time_ms = 0.0;

if (atomic)
time_ms = tt_scatter_atomic_wrapper(
dev_pattern, dev_sparse, dev_dense, pattern_length, delta, wrap, count);
else
time_ms = tt_scatter_wrapper(
dev_pattern, dev_sparse, dev_dense, pattern_length, delta, wrap, count);

if (time_ms < 0.0f) {
std::cerr << "Tenstorrent backend: scatter variant not implemented"
<< std::endl;
return;
}

if (timed)
time_seconds[run_id] = ((double)time_ms / 1000.0);
}

// gather_scatter / multi_gather / multi_scatter are declared so the class is
// concrete, but the corresponding wrappers return a negative sentinel in this
// draft. Neither is exercised by the stream or ustride suites.
void Configuration<Spatter::Tenstorrent>::gather_scatter(
bool timed, unsigned long run_id) {
(void)timed;
(void)run_id;
std::cerr << "Tenstorrent backend: gather_scatter not implemented"
<< std::endl;
}

void Configuration<Spatter::Tenstorrent>::multi_gather(
bool timed, unsigned long run_id) {
(void)timed;
(void)run_id;
std::cerr << "Tenstorrent backend: multi_gather not implemented" << std::endl;
}

void Configuration<Spatter::Tenstorrent>::multi_scatter(
bool timed, unsigned long run_id) {
(void)timed;
(void)run_id;
std::cerr << "Tenstorrent backend: multi_scatter not implemented"
<< std::endl;
}

void Configuration<Spatter::Tenstorrent>::setup() {
ConfigurationBase::setup();

// One page holds the whole pattern; the kernel reads page 0 once at start.
// Empty pattern vectors stay null -- a zero-byte allocation is an error.
auto upload = [](size_t *&handle, const aligned_vector<size_t> &p) {
if (p.empty())
return;
const size_t bytes = p.size() * sizeof(uint32_t);
handle = static_cast<size_t *>(tt_device_alloc(bytes, bytes));
tt_pattern_upload(handle, p.data(), p.size());
};

upload(dev_pattern, pattern);
upload(dev_pattern_gather, pattern_gather);
upload(dev_pattern_scatter, pattern_scatter);
}
#endif

} // namespace Spatter
Loading