Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,8 @@ struct TensorRTBlobHeader {
std::vector<AliasedBinding> aliased_io;
bool hardware_compatible = false;
int device_id = 0;
// Named-data key of an engine stored outside the blob; empty when it is inline.
std::string engine_key;

static const void* engine_data(const void* blob, const TensorRTBlobHeader& h);
static bool parse(const void* data, std::size_t size, TensorRTBlobHeader& out);
Expand Down
23 changes: 20 additions & 3 deletions cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
#include <map>
#include <memory>
#include <mutex>
#include <optional>
#include <string>
#include <tuple>
#include <utility>
Expand Down Expand Up @@ -1061,14 +1062,30 @@ Result<DelegateHandle*> TensorRTBackend::init(
}

const void* engine_data = TensorRTBlobHeader::engine_data(processed->data(), header);
std::size_t engine_size = header.engine_size;
std::optional<FreeableBuffer> named_engine;
if (!header.engine_key.empty()) {
const auto* named_data = context.get_named_data_map();
TORCHTRT_ET_CHECK_NOT_NULL(
named_data, Error::InvalidProgram, "TensorRTBackend::init: engine is named data but the program has none");
auto engine_buffer = named_data->get_data(header.engine_key);
if (!engine_buffer.ok()) {
ET_LOG(Error, "TensorRTBackend::init: named engine '%s' not found", header.engine_key.c_str());
return engine_buffer.error();
}
named_engine.emplace(std::move(engine_buffer.get()));
engine_data = named_engine->data();
engine_size = named_engine->size();
}
const bool share = !share_spec.ok() || share_spec.get();
Error err;
if (share) {
err =
acquire_shared_engine(*runtime, engine_data, header.engine_size, handle->device_id, ws_request, handle->engine);
err = acquire_shared_engine(*runtime, engine_data, engine_size, handle->device_id, ws_request, handle->engine);
} else {
err = load_engine(*runtime, engine_data, header.engine_size, ws_request, handle->engine);
err = load_engine(*runtime, engine_data, engine_size, ws_request, handle->engine);
}
// The engine now lives on the GPU, so release the host copy before building the context.
named_engine.reset();
if (err != Error::Ok) {
return err;
}
Expand Down
12 changes: 11 additions & 1 deletion cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -474,6 +474,15 @@ bool parse_metadata_json(const std::string& json, bool expects_aliased_io, Tenso
if (expects_aliased_io && out.aliased_io.empty()) {
return false;
}
out.engine_key.clear();
const std::size_t engine_key_pos = find_top_level_key(json, "\"engine_key\"");
if (engine_key_pos != std::string::npos) {
const std::size_t colon = json.find(':', engine_key_pos);
if (colon == std::string::npos ||
parse_string(json, skip_ws(json, colon + 1), out.engine_key) == std::string::npos || out.engine_key.empty()) {
return false;
}
}
const std::size_t hw_key = find_top_level_key(json, "\"hardware_compatible\"");
const std::size_t device_key = find_top_level_key(json, "\"device_id\"");
return parse_bool_after_key(json, hw_key, "\"hardware_compatible\"", out.hardware_compatible) &&
Expand Down Expand Up @@ -532,7 +541,8 @@ bool TensorRTBlobHeader::parse(const void* data, std::size_t size, TensorRTBlobH
}

std::string json(reinterpret_cast<const char*>(bytes + out.metadata_offset), out.metadata_size);
return parse_metadata_json(json, aliased_io_magic, out);
// An engine is either inline or named, never both, so neither copy can be silently ignored.
return parse_metadata_json(json, aliased_io_magic, out) && out.engine_key.empty() == (out.engine_size != 0);
}

} // namespace executorch_backend
Expand Down
24 changes: 21 additions & 3 deletions py/torch_tensorrt/executorch/_export_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@

import copy
import logging
import warnings
from typing import Any, Sequence

import torch
Expand Down Expand Up @@ -188,17 +189,34 @@ def _resolve_engine_tensor(exported_program: Any, node: Any) -> "torch.Tensor":
engine_obj = _resolve_engine_object(exported_program, node)
raw = getattr(engine_obj, "serialized_engine_tensor", None)
if raw is not None:
return raw()
# raw() views TensorRT's own buffer, and the returned tensor stays registered on
# the program until the save ends. ExecuTorch writes named data only from
# bytes, so copy now, while TensorRT's buffer is the only other copy.
return engine_bytes_tensor(raw().numpy().tobytes())
_warn_missing_accessor("serialized_engine_tensor")

engine_bytes = _resolve_engine_info(exported_program, node)[ENGINE_IDX]
if isinstance(engine_bytes, str):
import base64

engine_bytes = base64.b64decode(engine_bytes)
elif not isinstance(engine_bytes, (bytes, bytearray)):
elif not isinstance(engine_bytes, bytes):
engine_bytes = bytes(engine_bytes)
return torch.frombuffer(bytearray(engine_bytes), dtype=torch.uint8)
return engine_bytes_tensor(engine_bytes)


def engine_bytes_tensor(engine_bytes: bytes) -> "torch.Tensor":
"""A uint8 view of ``engine_bytes`` that remembers the ``bytes`` object.

ExecuTorch stores named data only as ``bytes``, so ``TensorRTBackend.preprocess``
reads ``_trt_engine_bytes`` back to hand over this same object rather than a copy.
Nothing writes an engine buffer, so viewing read-only memory is safe.
"""
with warnings.catch_warnings():
warnings.filterwarnings("ignore", message="The given buffer is not writable")
tensor = torch.frombuffer(engine_bytes, dtype=torch.uint8)
tensor._trt_engine_bytes = engine_bytes
return tensor


def validate_engine_program(
Expand Down
47 changes: 32 additions & 15 deletions py/torch_tensorrt/executorch/backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,12 +4,14 @@
# ExecuTorch TensorRT backend: serialize engines to a libtorch-free runtime blob.

import copy
import hashlib
import json
import operator
from typing import Any, Container, Iterable, List, Optional, Set, final

import torch
import torch.fx
from executorch.exir._serialize._named_data_store import NamedDataStore
from executorch.exir.backend.backend_details import (
BackendDetails,
CompileSpec,
Expand Down Expand Up @@ -525,19 +527,25 @@ def preprocess(
engine_info = list(engine_info)
_validate_engine_info(engine_info)
serialized_engine = engine_info[ENGINE_IDX]
if isinstance(serialized_engine, torch.Tensor):
# `bytes(storage)` looks equivalent but has two problems. It iterates
# the storage element by element in Python, costing about two seconds
# per megabyte, and it returns the whole backing allocation rather than
# the tensor's own extent, so a view of a larger buffer serializes too
# many bytes. A view and not a bytes copy, because serialize_engine
# already copies the engine into the blob once, and an engine can be
# several gigabytes. `.view(torch.uint8)` keeps `.numpy()` from
# rejecting a dtype it has no equivalent for.
engine_bytes = serialized_engine.cpu().contiguous().view(torch.uint8)
engine_info[ENGINE_IDX] = memoryview(engine_bytes.numpy())
elif not isinstance(serialized_engine, (bytes, bytearray)):
engine_info[ENGINE_IDX] = bytes(serialized_engine)
# ExecuTorch writes named data only from `bytes`, and an engine can be several
# gigabytes, so reuse the `bytes` the export staged the engine from
# (engine_bytes_tensor) and copy only a tensor that did not come from one.
engine_bytes = getattr(serialized_engine, "_trt_engine_bytes", None)
if not isinstance(engine_bytes, bytes):
if isinstance(serialized_engine, torch.Tensor):
# `bytes(storage)` looks equivalent but iterates element by element in
# Python, about two seconds per megabyte, and returns the whole backing
# allocation rather than the tensor's own extent. `.view(torch.uint8)`
# keeps `.numpy()` from rejecting a dtype it has no equivalent for.
engine_bytes = (
serialized_engine.cpu()
.contiguous()
.view(torch.uint8)
.numpy()
.tobytes()
)
else:
engine_bytes = bytes(serialized_engine)
input_names = _reorder_input_names_for_executorch(
edge_program,
engine_node,
Expand Down Expand Up @@ -587,6 +595,15 @@ def preprocess(
device_id=_parse_device_id(engine_info[DEVICE_IDX]),
serialized_metadata=_get_str(engine_info, SERIALIZED_METADATA_IDX),
target_platform=_get_str(engine_info, TARGET_PLATFORM_IDX),
# Content-keyed, so identical engines in different methods share one
# entry and different engines can never collide on a key.
engine_key="tensorrt_engine_" + hashlib.sha256(engine_bytes).hexdigest(),
)
# The engine goes to named data by reference and is streamed into the .pte,
# so the delegate blob holds only the header and metadata.
named_data = NamedDataStore()
named_data.add_named_data(metadata.engine_key, engine_bytes, alignment=16)
return PreprocessResult(
processed_bytes=serialize_engine(b"", metadata),
data_store_output=named_data.get_named_data_store_output(),
)
blob = serialize_engine(engine_info[ENGINE_IDX], metadata)
return PreprocessResult(processed_bytes=blob)
6 changes: 6 additions & 0 deletions py/torch_tensorrt/executorch/serialization.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,9 @@ class TensorRTBlobMetadata:
device_id: int = 0
serialized_metadata: str = ""
target_platform: str = ""
# Named-data key of the engine when it is stored outside the blob, which then
# carries an empty engine. Empty means the engine follows the metadata inline.
engine_key: str = ""

def to_json(self) -> bytes:
# Keep field order stable because the C++ parser is intentionally small.
Expand Down Expand Up @@ -80,6 +83,8 @@ def to_json(self) -> bytes:
"serialized_metadata": self.serialized_metadata,
"target_platform": self.target_platform,
}
if self.engine_key:
data["engine_key"] = self.engine_key
return json.dumps(data, separators=(",", ":")).encode("utf-8")

@classmethod
Expand All @@ -103,6 +108,7 @@ def from_json(cls, data: bytes) -> "TensorRTBlobMetadata":
device_id=parsed.get("device_id", 0),
serialized_metadata=parsed.get("serialized_metadata", ""),
target_platform=parsed.get("target_platform", ""),
engine_key=parsed.get("engine_key", ""),
)


Expand Down
17 changes: 17 additions & 0 deletions tests/cpp/executorch/test_executorch_blob_header.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -759,6 +759,23 @@ TEST(ExecuTorchTensorRTBlobHeader, RejectsUnknownFutureMagic) {
EXPECT_FALSE(TensorRTBlobHeader::parse(blob.data(), blob.size(), header));
}

TEST(ExecuTorchTensorRTBlobHeader, ReadsTheEngineKeyOfAnEngineStoredAsNamedData) {
const std::string bindings = R"({"io_bindings":[{"name":"x","is_input":true}],)";
const auto blob = make_blob(bindings + R"("engine_key":"tensorrt_engine_ab"})", 0);
TensorRTBlobHeader header;
ASSERT_TRUE(TensorRTBlobHeader::parse(blob.data(), blob.size(), header));
EXPECT_EQ(header.engine_key, "tensorrt_engine_ab");
EXPECT_EQ(header.engine_size, 0u);

// Inline and named at once leaves one of the two engines unread, and neither leaves none to load.
const auto both = make_blob(bindings + R"("engine_key":"tensorrt_engine_ab"})", 4);
EXPECT_FALSE(TensorRTBlobHeader::parse(both.data(), both.size(), header));
const auto neither = make_blob(VALID_METADATA, 0);
EXPECT_FALSE(TensorRTBlobHeader::parse(neither.data(), neither.size(), header));
const auto empty_key = make_blob(bindings + R"("engine_key":""})", 0);
EXPECT_FALSE(TensorRTBlobHeader::parse(empty_key.data(), empty_key.size(), header));
}

} // namespace
} // namespace executorch_backend
} // namespace torch_tensorrt
55 changes: 20 additions & 35 deletions tests/py/dynamo/executorch/test_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,9 +3,9 @@

import ast
import operator
import struct
from pathlib import Path
from types import SimpleNamespace
from typing import Any

import pytest

Expand All @@ -28,8 +28,6 @@
_get_engine_info_from_edge_program,
)
from torch_tensorrt.executorch.serialization import ( # noqa: E402
HEADER_FORMAT,
HEADER_SIZE,
TENSORRT_MAGIC,
deserialize_engine,
)
Expand Down Expand Up @@ -131,6 +129,14 @@ def _engine_tensor(payload: bytes) -> torch.Tensor:
return torch.frombuffer(bytearray(payload), dtype=torch.uint8)


def _named_engine(result: Any) -> bytes:
"""The engine a preprocess result stores as named data; its blob holds none."""
blob_engine, metadata = deserialize_engine(result.processed_bytes)
assert blob_engine == b""
store = result.data_store_output
return store.buffers[store.pte_data[metadata.engine_key].buffer_index]


@pytest.mark.unit
def test_no_op_placeholder_schema_matches_serialized_engine_layout():
meta_ops_path = (
Expand Down Expand Up @@ -191,8 +197,8 @@ def test_preprocess_serializes_engine_blob():

assert isinstance(result.processed_bytes, bytes)
assert result.processed_bytes[:4] == TENSORRT_MAGIC
engine, metadata = deserialize_engine(result.processed_bytes)
assert engine == b"engine-bytes"
assert _named_engine(result) == b"engine-bytes"
_, metadata = deserialize_engine(result.processed_bytes)
assert metadata.device_id == 2
assert [binding.name for binding in metadata.io_bindings] == ["x", "y"]
assert [binding.is_input for binding in metadata.io_bindings] == [True, False]
Expand All @@ -202,9 +208,6 @@ def test_preprocess_serializes_engine_blob():
def test_preprocess_serializes_only_the_engine_tensors_extent():
# A tensor that views part of a larger buffer. Serializing the whole storage
# instead of the tensor's own extent pads the engine with the trailing bytes.
# The recorded engine size is what exposes it: deserialize_engine trims the
# blob back to that size, so an over-long engine still round-trips and only
# the size gives it away.
payload = b"engine-bytes"
backing = torch.frombuffer(bytearray(payload + b"TRAILING"), dtype=torch.uint8)
engine_info = [""] * SERIALIZATION_LEN
Expand All @@ -216,12 +219,7 @@ def test_preprocess_serializes_only_the_engine_tensors_extent():

result = TensorRTBackend.preprocess(edge_program, [])

_, _, _, _, engine_size, _ = struct.unpack(
HEADER_FORMAT, result.processed_bytes[:HEADER_SIZE]
)
assert engine_size == len(payload)
engine, _ = deserialize_engine(result.processed_bytes)
assert engine == payload
assert _named_engine(result) == payload


@pytest.mark.unit
Expand All @@ -246,36 +244,23 @@ def forward(self, x):


@pytest.mark.unit
def test_preprocess_hands_the_engine_to_the_blob_without_copying_it(monkeypatch):
# The blob concatenation is the one copy of the engine preprocess needs. Turning
# the engine tensor into bytes first would hold one more copy at the same time,
# and an engine can be several gigabytes.
import numpy as np

from torch_tensorrt.executorch import backend as backend_module
def test_preprocess_hands_the_staged_engine_bytes_to_named_data_without_copying():
# An engine can be several gigabytes. The export stages it as a view of the
# module's own bytes, and the .pte is written from named data by reference, so
# the same object must come out the other end -- any copy is a whole engine more.
from torch_tensorrt.executorch._export_utils import engine_bytes_tensor

engine_tensor = _engine_tensor(b"engine-bytes")
payload = b"engine-bytes"
engine_info = [""] * SERIALIZATION_LEN
engine_info[ENGINE_IDX] = engine_tensor
engine_info[ENGINE_IDX] = engine_bytes_tensor(payload)
engine_info[DEVICE_IDX] = "0%8%0%0%GPU"
engine_info[INPUT_BINDING_NAMES_IDX] = "x"
engine_info[OUTPUT_BINDING_NAMES_IDX] = "y"
edge_program = _build_edge_program(engine_info)

received = []
real_serialize_engine = backend_module.serialize_engine

def recording_serialize_engine(engine_bytes, metadata):
received.append(engine_bytes)
return real_serialize_engine(engine_bytes, metadata)

monkeypatch.setattr(backend_module, "serialize_engine", recording_serialize_engine)
result = TensorRTBackend.preprocess(edge_program, [])

assert len(received) == 1
assert np.shares_memory(np.asarray(received[0]), engine_tensor.numpy())
engine, _ = deserialize_engine(result.processed_bytes)
assert engine == b"engine-bytes"
assert _named_engine(result) is payload


@pytest.mark.unit
Expand Down
Loading
Loading