diff --git a/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h b/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h index 4195c10da3..b42ba0c527 100644 --- a/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h +++ b/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h @@ -33,6 +33,8 @@ struct TensorRTBlobHeader { std::vector aliased_io; bool hardware_compatible = false; int device_id = 0; + // Named-data key of an engine stored outside the blob; empty when it is inline. + std::string engine_key; static const void* engine_data(const void* blob, const TensorRTBlobHeader& h); static bool parse(const void* data, std::size_t size, TensorRTBlobHeader& out); diff --git a/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp b/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp index 82a8c85b54..f3906e13fc 100644 --- a/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp +++ b/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp @@ -24,6 +24,7 @@ #include #include #include +#include #include #include #include @@ -1061,14 +1062,30 @@ Result TensorRTBackend::init( } const void* engine_data = TensorRTBlobHeader::engine_data(processed->data(), header); + std::size_t engine_size = header.engine_size; + std::optional named_engine; + if (!header.engine_key.empty()) { + const auto* named_data = context.get_named_data_map(); + TORCHTRT_ET_CHECK_NOT_NULL( + named_data, Error::InvalidProgram, "TensorRTBackend::init: engine is named data but the program has none"); + auto engine_buffer = named_data->get_data(header.engine_key); + if (!engine_buffer.ok()) { + ET_LOG(Error, "TensorRTBackend::init: named engine '%s' not found", header.engine_key.c_str()); + return engine_buffer.error(); + } + named_engine.emplace(std::move(engine_buffer.get())); + engine_data = named_engine->data(); + engine_size = named_engine->size(); + } const bool share = !share_spec.ok() || share_spec.get(); Error err; if (share) { - err = - acquire_shared_engine(*runtime, engine_data, header.engine_size, handle->device_id, ws_request, handle->engine); + err = acquire_shared_engine(*runtime, engine_data, engine_size, handle->device_id, ws_request, handle->engine); } else { - err = load_engine(*runtime, engine_data, header.engine_size, ws_request, handle->engine); + err = load_engine(*runtime, engine_data, engine_size, ws_request, handle->engine); } + // The engine now lives on the GPU, so release the host copy before building the context. + named_engine.reset(); if (err != Error::Ok) { return err; } diff --git a/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp b/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp index 24eb26ff94..5b1e458e85 100644 --- a/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp +++ b/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp @@ -474,6 +474,15 @@ bool parse_metadata_json(const std::string& json, bool expects_aliased_io, Tenso if (expects_aliased_io && out.aliased_io.empty()) { return false; } + out.engine_key.clear(); + const std::size_t engine_key_pos = find_top_level_key(json, "\"engine_key\""); + if (engine_key_pos != std::string::npos) { + const std::size_t colon = json.find(':', engine_key_pos); + if (colon == std::string::npos || + parse_string(json, skip_ws(json, colon + 1), out.engine_key) == std::string::npos || out.engine_key.empty()) { + return false; + } + } const std::size_t hw_key = find_top_level_key(json, "\"hardware_compatible\""); const std::size_t device_key = find_top_level_key(json, "\"device_id\""); return parse_bool_after_key(json, hw_key, "\"hardware_compatible\"", out.hardware_compatible) && @@ -532,7 +541,8 @@ bool TensorRTBlobHeader::parse(const void* data, std::size_t size, TensorRTBlobH } std::string json(reinterpret_cast(bytes + out.metadata_offset), out.metadata_size); - return parse_metadata_json(json, aliased_io_magic, out); + // An engine is either inline or named, never both, so neither copy can be silently ignored. + return parse_metadata_json(json, aliased_io_magic, out) && out.engine_key.empty() == (out.engine_size != 0); } } // namespace executorch_backend diff --git a/py/torch_tensorrt/executorch/_export_utils.py b/py/torch_tensorrt/executorch/_export_utils.py index 6d90cd3273..b3560a5fb1 100644 --- a/py/torch_tensorrt/executorch/_export_utils.py +++ b/py/torch_tensorrt/executorch/_export_utils.py @@ -5,6 +5,7 @@ import copy import logging +import warnings from typing import Any, Sequence import torch @@ -188,7 +189,10 @@ def _resolve_engine_tensor(exported_program: Any, node: Any) -> "torch.Tensor": engine_obj = _resolve_engine_object(exported_program, node) raw = getattr(engine_obj, "serialized_engine_tensor", None) if raw is not None: - return raw() + # raw() views TensorRT's own buffer, and the returned tensor stays registered on + # the program until the save ends. ExecuTorch writes named data only from + # bytes, so copy now, while TensorRT's buffer is the only other copy. + return engine_bytes_tensor(raw().numpy().tobytes()) _warn_missing_accessor("serialized_engine_tensor") engine_bytes = _resolve_engine_info(exported_program, node)[ENGINE_IDX] @@ -196,9 +200,23 @@ def _resolve_engine_tensor(exported_program: Any, node: Any) -> "torch.Tensor": import base64 engine_bytes = base64.b64decode(engine_bytes) - elif not isinstance(engine_bytes, (bytes, bytearray)): + elif not isinstance(engine_bytes, bytes): engine_bytes = bytes(engine_bytes) - return torch.frombuffer(bytearray(engine_bytes), dtype=torch.uint8) + return engine_bytes_tensor(engine_bytes) + + +def engine_bytes_tensor(engine_bytes: bytes) -> "torch.Tensor": + """A uint8 view of ``engine_bytes`` that remembers the ``bytes`` object. + + ExecuTorch stores named data only as ``bytes``, so ``TensorRTBackend.preprocess`` + reads ``_trt_engine_bytes`` back to hand over this same object rather than a copy. + Nothing writes an engine buffer, so viewing read-only memory is safe. + """ + with warnings.catch_warnings(): + warnings.filterwarnings("ignore", message="The given buffer is not writable") + tensor = torch.frombuffer(engine_bytes, dtype=torch.uint8) + tensor._trt_engine_bytes = engine_bytes + return tensor def validate_engine_program( diff --git a/py/torch_tensorrt/executorch/backend.py b/py/torch_tensorrt/executorch/backend.py index 65056cc4d6..dedc287a03 100644 --- a/py/torch_tensorrt/executorch/backend.py +++ b/py/torch_tensorrt/executorch/backend.py @@ -4,12 +4,14 @@ # ExecuTorch TensorRT backend: serialize engines to a libtorch-free runtime blob. import copy +import hashlib import json import operator from typing import Any, Container, Iterable, List, Optional, Set, final import torch import torch.fx +from executorch.exir._serialize._named_data_store import NamedDataStore from executorch.exir.backend.backend_details import ( BackendDetails, CompileSpec, @@ -525,19 +527,25 @@ def preprocess( engine_info = list(engine_info) _validate_engine_info(engine_info) serialized_engine = engine_info[ENGINE_IDX] - if isinstance(serialized_engine, torch.Tensor): - # `bytes(storage)` looks equivalent but has two problems. It iterates - # the storage element by element in Python, costing about two seconds - # per megabyte, and it returns the whole backing allocation rather than - # the tensor's own extent, so a view of a larger buffer serializes too - # many bytes. A view and not a bytes copy, because serialize_engine - # already copies the engine into the blob once, and an engine can be - # several gigabytes. `.view(torch.uint8)` keeps `.numpy()` from - # rejecting a dtype it has no equivalent for. - engine_bytes = serialized_engine.cpu().contiguous().view(torch.uint8) - engine_info[ENGINE_IDX] = memoryview(engine_bytes.numpy()) - elif not isinstance(serialized_engine, (bytes, bytearray)): - engine_info[ENGINE_IDX] = bytes(serialized_engine) + # ExecuTorch writes named data only from `bytes`, and an engine can be several + # gigabytes, so reuse the `bytes` the export staged the engine from + # (engine_bytes_tensor) and copy only a tensor that did not come from one. + engine_bytes = getattr(serialized_engine, "_trt_engine_bytes", None) + if not isinstance(engine_bytes, bytes): + if isinstance(serialized_engine, torch.Tensor): + # `bytes(storage)` looks equivalent but iterates element by element in + # Python, about two seconds per megabyte, and returns the whole backing + # allocation rather than the tensor's own extent. `.view(torch.uint8)` + # keeps `.numpy()` from rejecting a dtype it has no equivalent for. + engine_bytes = ( + serialized_engine.cpu() + .contiguous() + .view(torch.uint8) + .numpy() + .tobytes() + ) + else: + engine_bytes = bytes(serialized_engine) input_names = _reorder_input_names_for_executorch( edge_program, engine_node, @@ -587,6 +595,15 @@ def preprocess( device_id=_parse_device_id(engine_info[DEVICE_IDX]), serialized_metadata=_get_str(engine_info, SERIALIZED_METADATA_IDX), target_platform=_get_str(engine_info, TARGET_PLATFORM_IDX), + # Content-keyed, so identical engines in different methods share one + # entry and different engines can never collide on a key. + engine_key="tensorrt_engine_" + hashlib.sha256(engine_bytes).hexdigest(), + ) + # The engine goes to named data by reference and is streamed into the .pte, + # so the delegate blob holds only the header and metadata. + named_data = NamedDataStore() + named_data.add_named_data(metadata.engine_key, engine_bytes, alignment=16) + return PreprocessResult( + processed_bytes=serialize_engine(b"", metadata), + data_store_output=named_data.get_named_data_store_output(), ) - blob = serialize_engine(engine_info[ENGINE_IDX], metadata) - return PreprocessResult(processed_bytes=blob) diff --git a/py/torch_tensorrt/executorch/serialization.py b/py/torch_tensorrt/executorch/serialization.py index a8efab0ec4..9b8050b566 100644 --- a/py/torch_tensorrt/executorch/serialization.py +++ b/py/torch_tensorrt/executorch/serialization.py @@ -51,6 +51,9 @@ class TensorRTBlobMetadata: device_id: int = 0 serialized_metadata: str = "" target_platform: str = "" + # Named-data key of the engine when it is stored outside the blob, which then + # carries an empty engine. Empty means the engine follows the metadata inline. + engine_key: str = "" def to_json(self) -> bytes: # Keep field order stable because the C++ parser is intentionally small. @@ -80,6 +83,8 @@ def to_json(self) -> bytes: "serialized_metadata": self.serialized_metadata, "target_platform": self.target_platform, } + if self.engine_key: + data["engine_key"] = self.engine_key return json.dumps(data, separators=(",", ":")).encode("utf-8") @classmethod @@ -103,6 +108,7 @@ def from_json(cls, data: bytes) -> "TensorRTBlobMetadata": device_id=parsed.get("device_id", 0), serialized_metadata=parsed.get("serialized_metadata", ""), target_platform=parsed.get("target_platform", ""), + engine_key=parsed.get("engine_key", ""), ) diff --git a/tests/cpp/executorch/test_executorch_blob_header.cpp b/tests/cpp/executorch/test_executorch_blob_header.cpp index d26bda2b44..1d7d6607a4 100644 --- a/tests/cpp/executorch/test_executorch_blob_header.cpp +++ b/tests/cpp/executorch/test_executorch_blob_header.cpp @@ -759,6 +759,23 @@ TEST(ExecuTorchTensorRTBlobHeader, RejectsUnknownFutureMagic) { EXPECT_FALSE(TensorRTBlobHeader::parse(blob.data(), blob.size(), header)); } +TEST(ExecuTorchTensorRTBlobHeader, ReadsTheEngineKeyOfAnEngineStoredAsNamedData) { + const std::string bindings = R"({"io_bindings":[{"name":"x","is_input":true}],)"; + const auto blob = make_blob(bindings + R"("engine_key":"tensorrt_engine_ab"})", 0); + TensorRTBlobHeader header; + ASSERT_TRUE(TensorRTBlobHeader::parse(blob.data(), blob.size(), header)); + EXPECT_EQ(header.engine_key, "tensorrt_engine_ab"); + EXPECT_EQ(header.engine_size, 0u); + + // Inline and named at once leaves one of the two engines unread, and neither leaves none to load. + const auto both = make_blob(bindings + R"("engine_key":"tensorrt_engine_ab"})", 4); + EXPECT_FALSE(TensorRTBlobHeader::parse(both.data(), both.size(), header)); + const auto neither = make_blob(VALID_METADATA, 0); + EXPECT_FALSE(TensorRTBlobHeader::parse(neither.data(), neither.size(), header)); + const auto empty_key = make_blob(bindings + R"("engine_key":""})", 0); + EXPECT_FALSE(TensorRTBlobHeader::parse(empty_key.data(), empty_key.size(), header)); +} + } // namespace } // namespace executorch_backend } // namespace torch_tensorrt diff --git a/tests/py/dynamo/executorch/test_backend.py b/tests/py/dynamo/executorch/test_backend.py index 1cd8196493..4dc6e434ec 100644 --- a/tests/py/dynamo/executorch/test_backend.py +++ b/tests/py/dynamo/executorch/test_backend.py @@ -3,9 +3,9 @@ import ast import operator -import struct from pathlib import Path from types import SimpleNamespace +from typing import Any import pytest @@ -28,8 +28,6 @@ _get_engine_info_from_edge_program, ) from torch_tensorrt.executorch.serialization import ( # noqa: E402 - HEADER_FORMAT, - HEADER_SIZE, TENSORRT_MAGIC, deserialize_engine, ) @@ -131,6 +129,14 @@ def _engine_tensor(payload: bytes) -> torch.Tensor: return torch.frombuffer(bytearray(payload), dtype=torch.uint8) +def _named_engine(result: Any) -> bytes: + """The engine a preprocess result stores as named data; its blob holds none.""" + blob_engine, metadata = deserialize_engine(result.processed_bytes) + assert blob_engine == b"" + store = result.data_store_output + return store.buffers[store.pte_data[metadata.engine_key].buffer_index] + + @pytest.mark.unit def test_no_op_placeholder_schema_matches_serialized_engine_layout(): meta_ops_path = ( @@ -191,8 +197,8 @@ def test_preprocess_serializes_engine_blob(): assert isinstance(result.processed_bytes, bytes) assert result.processed_bytes[:4] == TENSORRT_MAGIC - engine, metadata = deserialize_engine(result.processed_bytes) - assert engine == b"engine-bytes" + assert _named_engine(result) == b"engine-bytes" + _, metadata = deserialize_engine(result.processed_bytes) assert metadata.device_id == 2 assert [binding.name for binding in metadata.io_bindings] == ["x", "y"] assert [binding.is_input for binding in metadata.io_bindings] == [True, False] @@ -202,9 +208,6 @@ def test_preprocess_serializes_engine_blob(): def test_preprocess_serializes_only_the_engine_tensors_extent(): # A tensor that views part of a larger buffer. Serializing the whole storage # instead of the tensor's own extent pads the engine with the trailing bytes. - # The recorded engine size is what exposes it: deserialize_engine trims the - # blob back to that size, so an over-long engine still round-trips and only - # the size gives it away. payload = b"engine-bytes" backing = torch.frombuffer(bytearray(payload + b"TRAILING"), dtype=torch.uint8) engine_info = [""] * SERIALIZATION_LEN @@ -216,12 +219,7 @@ def test_preprocess_serializes_only_the_engine_tensors_extent(): result = TensorRTBackend.preprocess(edge_program, []) - _, _, _, _, engine_size, _ = struct.unpack( - HEADER_FORMAT, result.processed_bytes[:HEADER_SIZE] - ) - assert engine_size == len(payload) - engine, _ = deserialize_engine(result.processed_bytes) - assert engine == payload + assert _named_engine(result) == payload @pytest.mark.unit @@ -246,36 +244,23 @@ def forward(self, x): @pytest.mark.unit -def test_preprocess_hands_the_engine_to_the_blob_without_copying_it(monkeypatch): - # The blob concatenation is the one copy of the engine preprocess needs. Turning - # the engine tensor into bytes first would hold one more copy at the same time, - # and an engine can be several gigabytes. - import numpy as np - - from torch_tensorrt.executorch import backend as backend_module +def test_preprocess_hands_the_staged_engine_bytes_to_named_data_without_copying(): + # An engine can be several gigabytes. The export stages it as a view of the + # module's own bytes, and the .pte is written from named data by reference, so + # the same object must come out the other end -- any copy is a whole engine more. + from torch_tensorrt.executorch._export_utils import engine_bytes_tensor - engine_tensor = _engine_tensor(b"engine-bytes") + payload = b"engine-bytes" engine_info = [""] * SERIALIZATION_LEN - engine_info[ENGINE_IDX] = engine_tensor + engine_info[ENGINE_IDX] = engine_bytes_tensor(payload) engine_info[DEVICE_IDX] = "0%8%0%0%GPU" engine_info[INPUT_BINDING_NAMES_IDX] = "x" engine_info[OUTPUT_BINDING_NAMES_IDX] = "y" edge_program = _build_edge_program(engine_info) - received = [] - real_serialize_engine = backend_module.serialize_engine - - def recording_serialize_engine(engine_bytes, metadata): - received.append(engine_bytes) - return real_serialize_engine(engine_bytes, metadata) - - monkeypatch.setattr(backend_module, "serialize_engine", recording_serialize_engine) result = TensorRTBackend.preprocess(edge_program, []) - assert len(received) == 1 - assert np.shares_memory(np.asarray(received[0]), engine_tensor.numpy()) - engine, _ = deserialize_engine(result.processed_bytes) - assert engine == b"engine-bytes" + assert _named_engine(result) is payload @pytest.mark.unit diff --git a/tests/py/dynamo/executorch/test_engine_info_accessors.py b/tests/py/dynamo/executorch/test_engine_info_accessors.py index 98ecf8b14c..6b77f59799 100644 --- a/tests/py/dynamo/executorch/test_engine_info_accessors.py +++ b/tests/py/dynamo/executorch/test_engine_info_accessors.py @@ -74,6 +74,13 @@ def __getstate__(self): return (_record(base64.b64encode(ENGINE_BYTES).decode()),) +class RuntimeWithTensorAccessor(RuntimeWithAccessor): + """Engine from a runtime that serializes into a buffer it owns, like the C++ one.""" + + def serialized_engine_tensor(self): + return torch.frombuffer(bytearray(ENGINE_BYTES), dtype=torch.uint8) + + class _Program: """Stand-in for ExportedProgram carrying what the two passes touch.""" @@ -185,3 +192,26 @@ def test_metadata_only_record_is_not_cross_served_as_engine_bytes(): "metadata-only record cached by validation rather than re-resolving them" ) assert engine.getstate_calls == 1, "the engine record still must be read once" + + +@pytest.mark.unit +def test_rewrite_stages_the_engine_once_as_bytes(): + """The buffer the rewrite registers lives until the save ends, and ExecuTorch + writes named data only from ``bytes``. A buffer still viewing the runtime's + serialization makes preprocess copy it, holding every engine twice.""" + import numpy as np + + program, _ = _program_with_engine(RuntimeWithTensorAccessor()) + _export_utils.replace_execute_engine(program, {}) + + no_op = torch.ops.tensorrt.no_op_placeholder_for_execute_engine.default + (no_op_node,) = [n for n in program.graph_module.graph.nodes if n.target is no_op] + engine_buffer = getattr( + program.graph_module, no_op_node.args[1 + ENGINE_IDX].target + ) + + engine_bytes = engine_buffer._trt_engine_bytes + assert engine_bytes == ENGINE_BYTES + assert np.shares_memory( + engine_buffer.numpy(), np.frombuffer(engine_bytes, dtype=np.uint8) + ), "the registered buffer must view the bytes preprocess hands to named data"