包裝外部張量記憶體
| 欄位 | 值 |
|---|---|
| 類別 | PCIe 協同處理 |
| 難度 | 中級 |
| 預估閱讀時間 | 15 分鐘 |
| 標籤 | PCIe, C++, tensor, external memory, zero-copy wrapping |
程式會建立三個可重複使用的輸入槽位,並提交八個合成 FP32 畫面。只有在提取該槽位依序對應的結果後,槽位才會回到可用佇列。
操作指南
檢查模型契約
先建構模型並讀取 info().inputs,再配置記憶體。本教學使用單一輸入的 YOLOv8s 封存檔,並確認回報的 dtype 是 FP32,而且其形狀正好占用 size_bytes 位元組。
外部檢視必須符合對應 TensorInfo 的 dtype、形狀、位元組大小和名稱。請勿從其他模型組建推導這些值。
pcie::ConnectionOptions connection;
connection.card_id = card_id;
connection.max_inflight = static_cast<int>(kRingSlots);
pcie::Model model(kModelPath, {}, connection);
const pcie::ModelInfo info = model.info();
if (info.inputs.size() != 1U) {
throw std::runtime_error("tutorial requires a model with one input");
}
validate_fp32_input(info.inputs.front());
包裝應用程式擁有的記憶體
每個環狀槽位都擁有 std::shared_ptr<std::vector<float>>。呼叫 Tensor::from_external() 時會傳入基底指標、完整的背後元素數、共享擁有者、模型形狀和路由名稱。因為檢視是連續的,PCIe 主機可直接包裝,而不建立暫存配置。
只保留原始指標並不足夠。傳輸可能在 push() 傳回後繼續保留張量,因此共享擁有者是必要的。
std::vector<InputSlot> slots = make_input_ring(info.inputs.front());
組建模型
驗證模型契約和環狀配置後再組建。本範例將 max_inflight 設為環狀大小,讓應用程式和傳輸具有相同的明確上限。
model.build(kBuildTimeoutMs);
提交並安全地重複使用環狀緩衝區
填入可用槽位、呼叫 push(),再將該槽位移至處理中佇列。請勿只因 push() 已傳回就修改或重複使用其儲存空間。沒有可用槽位時,範例會呼叫 pull(),並只在相符的依序結果抵達後,才將最舊槽位送回可用佇列。
每個已接受的推送都對應一次提取,包括最後的清空。若逾時,程式會關閉模型,而不重複使用要求仍可能有效的記憶體。
std::size_t completed = 0;
try {
completed = run_input_ring(model, slots);
} catch (...) {
model.close();
throw;
}
model.close();
執行
請依照教學設定的說明安裝 PCIe 主機套件並下載教學套件組。將 YOLOv8s 下載到解壓縮後的 PCIe extras 根目錄:
sima-cli modelzoo get yolo_v8s
cp /absolute/path/to/downloaded-yolov8s-archive.tar.gz yolo_v8s_mpk.tar.gz
test -f yolo_v8s_mpk.tar.gz
C++ (prebuilt):
./lib/sima-pcie-host/tutorials/tutorial_028_wrap_external_tensor_memory
C++ (build from source):
./build.sh --target tutorial_028_wrap_external_tensor_memory
./build/tutorials-standalone/tutorial_028_wrap_external_tensor_memory
預設使用卡片 0 和佇列 0。只有使用其他卡片時才傳入 --card N。成功執行會列印:
input=images
ring_slots=3
completed=8
[OK] 028_wrap_external_tensor_memory
實務應用
直接主機包裝路徑需要連續儲存空間。只要描述元有效,具有非連續步幅的張量仍可接受,但主機會將它壓實到暫存配置。多個分別配置的輸入也會封裝到暫存記憶體。
對於多輸入模型,只有當所有張量都是同一個共享封裝配置中的連續檢視時,才能避免這個暫存配置。請依 info().inputs 回報的順序提交,並使用每個輸入的名稱和形狀。例如,當兩個輸入都是 FP32 時:
const auto& first = info.inputs.at(0);
const auto& second = info.inputs.at(1);
const std::size_t first_count = first.size_bytes / sizeof(float);
const std::size_t second_count = second.size_bytes / sizeof(float);
auto packed =
std::make_shared<std::vector<float>>(first_count + second_count);
pcie::Tensor input0 = pcie::Tensor::from_external(
packed->data(), packed->size(), packed, first.shape, first.name);
pcie::Tensor input1 = pcie::Tensor::from_external(
packed->data(), packed->size(), packed, second.shape, second.name,
static_cast<std::int64_t>(first.size_bytes));
model.push({input0, input1});
這項最佳化只會移除主機端封裝複本。PCIe 在推論前仍會將封裝的承載資料複製到卡片擁有的傳輸記憶體。
完整原始碼
顯示完整原始碼程式
// Submit application-owned tensor memory without a host staging copy.
//
// Usage:
// tutorial_028_wrap_external_tensor_memory [--card 0]
#include <simaai/neat/pcie/Model.h>
#include <algorithm>
#include <cstdint>
#include <cstdlib>
#include <deque>
#include <filesystem>
#include <iostream>
#include <limits>
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
namespace pcie = simaai::neat::pcie;
namespace {
constexpr int kBuildTimeoutMs = 180000;
constexpr int kPullTimeoutMs = 30000;
constexpr std::size_t kRingSlots = 3;
constexpr std::size_t kFrameCount = 8;
constexpr char kModelPath[] = "yolo_v8s_mpk.tar.gz";
int parse_card(const int argc, char** argv) {
int card_id = 0;
for (int index = 1; index < argc; ++index) {
const std::string arg = argv[index];
if (arg == "--card" && index + 1 < argc) {
card_id = std::stoi(argv[++index]);
} else if (arg == "-h" || arg == "--help") {
std::cout << "Usage: " << argv[0] << " [--card 0]\n";
std::exit(0);
} else {
throw std::runtime_error("unknown or incomplete argument: " + arg);
}
}
return card_id;
}
std::size_t checked_element_count(const std::vector<std::int64_t>& shape) {
std::size_t count = 1;
for (const std::int64_t dimension : shape) {
if (dimension <= 0 ||
count > std::numeric_limits<std::size_t>::max() / static_cast<std::size_t>(dimension)) {
throw std::runtime_error("model input has an invalid or overflowing shape");
}
count *= static_cast<std::size_t>(dimension);
}
return count;
}
void validate_fp32_input(const pcie::TensorInfo& input) {
if (input.dtype != "FP32" && input.dtype != "FLOAT32") {
throw std::runtime_error("tutorial requires an FP32 model input, got " + input.dtype);
}
const std::size_t element_count = checked_element_count(input.shape);
if (element_count > std::numeric_limits<std::size_t>::max() / sizeof(float) ||
element_count * sizeof(float) != input.size_bytes) {
throw std::runtime_error("model input shape and byte size are inconsistent");
}
}
struct InputSlot {
std::shared_ptr<std::vector<float>> storage;
pcie::Tensor tensor;
};
InputSlot make_input_slot(const pcie::TensorInfo& input) {
auto storage = std::make_shared<std::vector<float>>(checked_element_count(input.shape), 0.0F);
pcie::Tensor tensor = pcie::Tensor::from_external(storage->data(), storage->size(), storage,
input.shape, input.name);
return {.storage = std::move(storage), .tensor = std::move(tensor)};
}
std::vector<InputSlot> make_input_ring(const pcie::TensorInfo& input) {
std::vector<InputSlot> slots;
slots.reserve(kRingSlots);
for (std::size_t index = 0; index < kRingSlots; ++index) {
slots.push_back(make_input_slot(input));
}
return slots;
}
std::size_t run_input_ring(pcie::Model& model, std::vector<InputSlot>& slots) {
std::deque<std::size_t> available;
std::deque<std::size_t> in_flight;
for (std::size_t index = 0; index < slots.size(); ++index) {
available.push_back(index);
}
std::size_t completed = 0;
const auto complete_oldest = [&] {
auto outputs = model.pull(kPullTimeoutMs);
if (!outputs) {
throw std::runtime_error("timed out waiting for an external-memory submission");
}
if (outputs->empty() || in_flight.empty()) {
throw std::runtime_error("received an invalid external-memory completion");
}
available.push_back(in_flight.front());
in_flight.pop_front();
++completed;
};
for (std::size_t frame = 0; frame < kFrameCount; ++frame) {
if (available.empty()) {
complete_oldest();
}
const std::size_t slot_index = available.front();
available.pop_front();
InputSlot& slot = slots[slot_index];
// A slot is writable only while it is not in flight.
std::fill(slot.storage->begin(), slot.storage->end(), static_cast<float>(frame % 10U) / 10.0F);
if (!model.push(slot.tensor)) {
throw std::runtime_error("push rejected frame " + std::to_string(frame));
}
in_flight.push_back(slot_index);
}
while (!in_flight.empty()) {
complete_oldest();
}
return completed;
}
} // namespace
int main(int argc, char** argv) {
try {
const int card_id = parse_card(argc, argv);
if (!std::filesystem::is_regular_file(kModelPath)) {
throw std::runtime_error(std::string("model does not exist: ") + kModelPath);
}
pcie::ConnectionOptions connection;
connection.card_id = card_id;
connection.max_inflight = static_cast<int>(kRingSlots);
pcie::Model model(kModelPath, {}, connection);
const pcie::ModelInfo info = model.info();
if (info.inputs.size() != 1U) {
throw std::runtime_error("tutorial requires a model with one input");
}
validate_fp32_input(info.inputs.front());
std::vector<InputSlot> slots = make_input_ring(info.inputs.front());
model.build(kBuildTimeoutMs);
std::size_t completed = 0;
try {
completed = run_input_ring(model, slots);
} catch (...) {
model.close();
throw;
}
model.close();
std::cout << "input=" << info.inputs.front().name << '\n'
<< "ring_slots=" << slots.size() << '\n'
<< "completed=" << completed << '\n'
<< "[OK] 028_wrap_external_tensor_memory\n";
return 0;
} catch (const std::exception& error) {
std::cerr << "[FAIL] " << error.what() << '\n';
return 1;
}
}