Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions .gitmodules
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
[submodule "third-party/fmt"]
path = third-party/fmt
url = https://github.com/fmtlib/fmt.git
[submodule "third-party/deep_jit"]
path = third-party/deep_jit
url = git@github.com:deepseek-ai/DeepJIT.git

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🔴 critical: 子模块 URL 使用 SSH 地址 git@github.com:deepseek-ai/DeepJIT.git。README 要求用户执行 git submodule update --init --recursive,未配置 GitHub SSH key 的匿名/HTTPS 用户会直接失败(原 fmt 子模块使用的是 HTTPS)。建议改为 https://github.com/deepseek-ai/DeepJIT.git。

🤖 v5

20 changes: 7 additions & 13 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
# NOTES: this CMake is only for debugging; for setup, please use Torch extension
cmake_minimum_required(VERSION 3.10)
cmake_minimum_required(VERSION 3.18)
project(deep_ep LANGUAGES CUDA CXX)
set(CMAKE_VERBOSE_MAKEFILE ON)

Expand All @@ -11,8 +11,6 @@ list(APPEND CUDA_NVCC_FLAGS "-O3")
list(APPEND CUDA_NVCC_FLAGS "--ptxas-options=--verbose,--register-usage-level=10,--warn-on-local-memory-usage")
list(APPEND CUDA_NVCC_FLAGS "-Xcompiler=-rdynamic")
list(APPEND CUDA_NVCC_FLAGS "-lineinfo")
# Suppress warnings for the `fmt` library
list(APPEND CUDA_NVCC_FLAGS "--diag-suppress=128,2417")

set(USE_SYSTEM_NVTX on)
set(CUDA_ARCH_LIST "9.0" CACHE STRING "List of CUDA architectures to compile")
Expand All @@ -22,31 +20,27 @@ find_package(CUDAToolkit REQUIRED)
find_package(pybind11 REQUIRED)
find_package(Torch REQUIRED)

# NVSHMEM
find_package(NVSHMEM REQUIRED HINTS ${NVSHMEM_ROOT_DIR}/lib/cmake/nvshmem)
add_library(nvshmem ALIAS nvshmem::nvshmem)
add_library(nvshmem_host ALIAS nvshmem::nvshmem_host)
add_library(nvshmem_device ALIAS nvshmem::nvshmem_device)

# NCCL
# TODO: use `find_package` instead of manual checks
if (NOT NCCL_ROOT_DIR)
message(FATAL_ERROR "NCCL_ROOT_DIR is not set.")
endif()

set(CMAKE_CXX_STANDARD 20)
set(CMAKE_CXX_STANDARD_REQUIRED ON)
set(CMAKE_CUDA_STANDARD 20)
set(CMAKE_CUDA_STANDARD_REQUIRED ON)

include_directories(deep_ep/include third-party/fmt/include)
include_directories(${CUDA_TOOLKIT_ROOT_DIR}/include ${CUDA_TOOLKIT_ROOT_DIR}/include/cccl ${TORCH_INCLUDE_DIRS} ${PYTHON_INCLUDE_DIRS} ${NVSHMEM_INCLUDE_DIR} ${NCCL_ROOT_DIR}/include .)
link_directories(${TORCH_INSTALL_PREFIX}/lib ${CUDA_TOOLKIT_ROOT_DIR}/lib ${NVSHMEM_LIB_DIR} ${NCCL_ROOT_DIR}/lib)
include_directories(deep_ep/include third-party/deep_jit/include)
include_directories(${CUDA_TOOLKIT_ROOT_DIR}/include ${CUDA_TOOLKIT_ROOT_DIR}/include/cccl ${TORCH_INCLUDE_DIRS} ${PYTHON_INCLUDE_DIRS} ${NCCL_ROOT_DIR}/include .)
link_directories(${TORCH_INSTALL_PREFIX}/lib ${CUDA_TOOLKIT_ROOT_DIR}/lib ${NCCL_ROOT_DIR}/lib)

# Add kernels
add_subdirectory(csrc)

# Link CPP and CUDA together
pybind11_add_module(_C csrc/python_api.cpp)
target_link_libraries(_C PRIVATE nccl ${RUNTIME_CUDA_LIBRARIES} ${LEGACY_CUDA_LIBRARIES} ${TORCH_LIBRARIES} torch_python)
target_link_libraries(_C PRIVATE nccl ${RUNTIME_CUDA_LIBRARIES} ${TORCH_LIBRARIES} torch_python)

# Enable kernel code indexing with CMake-based IDEs
cuda_add_library(deep_ep_indexing_cuda STATIC csrc/indexing/main.cu)
464 changes: 272 additions & 192 deletions README.md

Large diffs are not rendered by default.

63 changes: 63 additions & 0 deletions csrc/buffers/base.hpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,63 @@
#pragma once

#include <cstdio>
#include <memory>
#include <pybind11/pybind11.h>

#include "../kernels/comm/api.hpp"

namespace deep_ep {

class BufferBase {
protected:
bool explicitly_destroy;
bool destroyed = false;

explicit BufferBase(const bool& explicitly_destroy): explicitly_destroy(explicitly_destroy) {}

// Called from the derived destructor, while its members and destroy() implementation are still available.
void destroy_on_destruction(const char* name) {
if (not explicitly_destroy and not destroyed) {
destroy();
main_context = nullptr;
}

if (not destroyed) {
std::printf("`destroy()` is not called before DeepEP %s buffer destruction, which can leak resources.\n", name);
std::fflush(stdout);
}
}

public:
std::shared_ptr<comm::Context> main_context;

virtual ~BufferBase() noexcept(false) = default;
virtual void destroy() = 0;

void print_memory_usage() const {
EP_HOST_ASSERT(not destroyed and main_context != nullptr);
const auto num_workspace_gibs = main_context->num_workspace_bytes / float(1 << 30);
const auto num_buffer_gibs = main_context->num_gpu_buffer_bytes / float(1 << 30);
const auto num_gpu_storage_gibs = main_context->use_cpu_rdma_storage ? 0.0 : main_context->num_rdma_storage_bytes / float(1 << 30);
const auto num_cpu_storage_gibs = main_context->use_cpu_rdma_storage ? main_context->num_rdma_storage_bytes / float(1 << 30) : 0.0;
std::printf("[DeepEP memory usage] rank_idx=%d, num_ranks=%d, GPU total %.3f GiB "
"(workspace %.3f GiB, buffer %.3f GiB, RDMA storage %.3f GiB), CPU total %.3f GiB\n",
main_context->rank_idx, main_context->num_ranks,
num_workspace_gibs + num_buffer_gibs + num_gpu_storage_gibs,
num_workspace_gibs, num_buffer_gibs, num_gpu_storage_gibs, num_cpu_storage_gibs);
std::fflush(stdout);
}
};

} // namespace deep_ep

namespace deep_ep::base {

static void register_apis(pybind11::module_& m) {
pybind11::class_<BufferBase>(m, "BufferBase")
.def_readonly("main_context", &BufferBase::main_context)
.def("print_memory_usage", &BufferBase::print_memory_usage)
.def("destroy", &BufferBase::destroy);
}

} // namespace deep_ep::base
Loading
Loading