From 659c9fdfb05cdc6718c8224fe754d7a4bfea5597 Mon Sep 17 00:00:00 2001 From: cehongwang Date: Tue, 6 Oct 2026 23:22:54 +0000 Subject: [PATCH] perf(executorch): store each engine once, as named data, when saving a .pte Saving a .pte kept two host copies of every engine until the file was written: the serialized plan, registered on the exported program as the engine's buffer, and the delegate blob, which preprocess built by copying the plan in after the header. With the resource partitioner the save then held two copies of all engines at once and became the export's peak. preprocess now hands the engine to ExecuTorch as named data, keyed by its SHA-256, and the blob carries only the header and metadata, with an engine_key naming the entry. ExecuTorch stores named data only as bytes, so _resolve_engine_tensor converts TensorRT's serialized plan to bytes once, while TensorRT's buffer is the only other copy, and preprocess reuses that same object. Identical engines in different methods share one entry. At load, TensorRTBackend::init reads the engine from the program's named data map when the header names one, passes it to the shared-engine cache or load_engine as before, and frees the host copy once the engine is on the GPU. Inline blobs load as before. The header parser rejects a blob that has both or neither. On pi0.5 (bf16, 9 engines at cpu_memory_budget=8 GiB, on top of the getitem fix in the resource partitioner), the save's peak host RSS drops from 22.1 to 14.0 GiB, which is the export's peak. --- .../executorch/TensorRTBlobHeader.h | 2 + .../executorch/TensorRTBackend.cpp | 23 +++++++- .../executorch/TensorRTBlobHeader.cpp | 12 +++- py/torch_tensorrt/executorch/_export_utils.py | 24 +++++++- py/torch_tensorrt/executorch/backend.py | 47 +++++++++++----- py/torch_tensorrt/executorch/serialization.py | 6 ++ .../test_executorch_blob_header.cpp | 17 ++++++ tests/py/dynamo/executorch/test_backend.py | 55 +++++++------------ .../executorch/test_engine_info_accessors.py | 30 ++++++++++ 9 files changed, 159 insertions(+), 57 deletions(-) diff --git a/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h b/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h index 4195c10da37..b42ba0c527a 100644 --- a/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h +++ b/cpp/include/torch_tensorrt/executorch/TensorRTBlobHeader.h @@ -33,6 +33,8 @@ struct TensorRTBlobHeader { std::vector aliased_io; bool hardware_compatible = false; int device_id = 0; + // Named-data key of an engine stored outside the blob; empty when it is inline. + std::string engine_key; static const void* engine_data(const void* blob, const TensorRTBlobHeader& h); static bool parse(const void* data, std::size_t size, TensorRTBlobHeader& out); diff --git a/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp b/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp index 82a8c85b544..f3906e13fc1 100644 --- a/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp +++ b/cpp/src/torch_tensorrt/executorch/TensorRTBackend.cpp @@ -24,6 +24,7 @@ #include #include #include +#include #include #include #include @@ -1061,14 +1062,30 @@ Result TensorRTBackend::init( } const void* engine_data = TensorRTBlobHeader::engine_data(processed->data(), header); + std::size_t engine_size = header.engine_size; + std::optional named_engine; + if (!header.engine_key.empty()) { + const auto* named_data = context.get_named_data_map(); + TORCHTRT_ET_CHECK_NOT_NULL( + named_data, Error::InvalidProgram, "TensorRTBackend::init: engine is named data but the program has none"); + auto engine_buffer = named_data->get_data(header.engine_key); + if (!engine_buffer.ok()) { + ET_LOG(Error, "TensorRTBackend::init: named engine '%s' not found", header.engine_key.c_str()); + return engine_buffer.error(); + } + named_engine.emplace(std::move(engine_buffer.get())); + engine_data = named_engine->data(); + engine_size = named_engine->size(); + } const bool share = !share_spec.ok() || share_spec.get(); Error err; if (share) { - err = - acquire_shared_engine(*runtime, engine_data, header.engine_size, handle->device_id, ws_request, handle->engine); + err = acquire_shared_engine(*runtime, engine_data, engine_size, handle->device_id, ws_request, handle->engine); } else { - err = load_engine(*runtime, engine_data, header.engine_size, ws_request, handle->engine); + err = load_engine(*runtime, engine_data, engine_size, ws_request, handle->engine); } + // The engine now lives on the GPU, so release the host copy before building the context. + named_engine.reset(); if (err != Error::Ok) { return err; } diff --git a/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp b/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp index 24eb26ff94a..5b1e458e859 100644 --- a/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp +++ b/cpp/src/torch_tensorrt/executorch/TensorRTBlobHeader.cpp @@ -474,6 +474,15 @@ bool parse_metadata_json(const std::string& json, bool expects_aliased_io, Tenso if (expects_aliased_io && out.aliased_io.empty()) { return false; } + out.engine_key.clear(); + const std::size_t engine_key_pos = find_top_level_key(json, "\"engine_key\""); + if (engine_key_pos != std::string::npos) { + const std::size_t colon = json.find(':', engine_key_pos); + if (colon == std::string::npos || + parse_string(json, skip_ws(json, colon + 1), out.engine_key) == std::string::npos || out.engine_key.empty()) { + return false; + } + } const std::size_t hw_key = find_top_level_key(json, "\"hardware_compatible\""); const std::size_t device_key = find_top_level_key(json, "\"device_id\""); return parse_bool_after_key(json, hw_key, "\"hardware_compatible\"", out.hardware_compatible) && @@ -532,7 +541,8 @@ bool TensorRTBlobHeader::parse(const void* data, std::size_t size, TensorRTBlobH } std::string json(reinterpret_cast(bytes + out.metadata_offset), out.metadata_size); - return parse_metadata_json(json, aliased_io_magic, out); + // An engine is either inline or named, never both, so neither copy can be silently ignored. + return parse_metadata_json(json, aliased_io_magic, out) && out.engine_key.empty() == (out.engine_size != 0); } } // namespace executorch_backend diff --git a/py/torch_tensorrt/executorch/_export_utils.py b/py/torch_tensorrt/executorch/_export_utils.py index 6d90cd3273f..b3560a5fb15 100644 --- a/py/torch_tensorrt/executorch/_export_utils.py +++ b/py/torch_tensorrt/executorch/_export_utils.py @@ -5,6 +5,7 @@ import copy import logging +import warnings from typing import Any, Sequence import torch @@ -188,7 +189,10 @@ def _resolve_engine_tensor(exported_program: Any, node: Any) -> "torch.Tensor": engine_obj = _resolve_engine_object(exported_program, node) raw = getattr(engine_obj, "serialized_engine_tensor", None) if raw is not None: - return raw() + # raw() views TensorRT's own buffer, and the returned tensor stays registered on + # the program until the save ends. ExecuTorch writes named data only from + # bytes, so copy now, while TensorRT's buffer is the only other copy. + return engine_bytes_tensor(raw().numpy().tobytes()) _warn_missing_accessor("serialized_engine_tensor") engine_bytes = _resolve_engine_info(exported_program, node)[ENGINE_IDX] @@ -196,9 +200,23 @@ def _resolve_engine_tensor(exported_program: Any, node: Any) -> "torch.Tensor": import base64 engine_bytes = base64.b64decode(engine_bytes) - elif not isinstance(engine_bytes, (bytes, bytearray)): + elif not isinstance(engine_bytes, bytes): engine_bytes = bytes(engine_bytes) - return torch.frombuffer(bytearray(engine_bytes), dtype=torch.uint8) + return engine_bytes_tensor(engine_bytes) + + +def engine_bytes_tensor(engine_bytes: bytes) -> "torch.Tensor": + """A uint8 view of ``engine_bytes`` that remembers the ``bytes`` object. + + ExecuTorch stores named data only as ``bytes``, so ``TensorRTBackend.preprocess`` + reads ``_trt_engine_bytes`` back to hand over this same object rather than a copy. + Nothing writes an engine buffer, so viewing read-only memory is safe. + """ + with warnings.catch_warnings(): + warnings.filterwarnings("ignore", message="The given buffer is not writable") + tensor = torch.frombuffer(engine_bytes, dtype=torch.uint8) + tensor._trt_engine_bytes = engine_bytes + return tensor def validate_engine_program( diff --git a/py/torch_tensorrt/executorch/backend.py b/py/torch_tensorrt/executorch/backend.py index 65056cc4d62..dedc287a036 100644 --- a/py/torch_tensorrt/executorch/backend.py +++ b/py/torch_tensorrt/executorch/backend.py @@ -4,12 +4,14 @@ # ExecuTorch TensorRT backend: serialize engines to a libtorch-free runtime blob. import copy +import hashlib import json import operator from typing import Any, Container, Iterable, List, Optional, Set, final import torch import torch.fx +from executorch.exir._serialize._named_data_store import NamedDataStore from executorch.exir.backend.backend_details import ( BackendDetails, CompileSpec, @@ -525,19 +527,25 @@ def preprocess( engine_info = list(engine_info) _validate_engine_info(engine_info) serialized_engine = engine_info[ENGINE_IDX] - if isinstance(serialized_engine, torch.Tensor): - # `bytes(storage)` looks equivalent but has two problems. It iterates - # the storage element by element in Python, costing about two seconds - # per megabyte, and it returns the whole backing allocation rather than - # the tensor's own extent, so a view of a larger buffer serializes too - # many bytes. A view and not a bytes copy, because serialize_engine - # already copies the engine into the blob once, and an engine can be - # several gigabytes. `.view(torch.uint8)` keeps `.numpy()` from - # rejecting a dtype it has no equivalent for. - engine_bytes = serialized_engine.cpu().contiguous().view(torch.uint8) - engine_info[ENGINE_IDX] = memoryview(engine_bytes.numpy()) - elif not isinstance(serialized_engine, (bytes, bytearray)): - engine_info[ENGINE_IDX] = bytes(serialized_engine) + # ExecuTorch writes named data only from `bytes`, and an engine can be several + # gigabytes, so reuse the `bytes` the export staged the engine from + # (engine_bytes_tensor) and copy only a tensor that did not come from one. + engine_bytes = getattr(serialized_engine, "_trt_engine_bytes", None) + if not isinstance(engine_bytes, bytes): + if isinstance(serialized_engine, torch.Tensor): + # `bytes(storage)` looks equivalent but iterates element by element in + # Python, about two seconds per megabyte, and returns the whole backing + # allocation rather than the tensor's own extent. `.view(torch.uint8)` + # keeps `.numpy()` from rejecting a dtype it has no equivalent for. + engine_bytes = ( + serialized_engine.cpu() + .contiguous() + .view(torch.uint8) + .numpy() + .tobytes() + ) + else: + engine_bytes = bytes(serialized_engine) input_names = _reorder_input_names_for_executorch( edge_program, engine_node, @@ -587,6 +595,15 @@ def preprocess( device_id=_parse_device_id(engine_info[DEVICE_IDX]), serialized_metadata=_get_str(engine_info, SERIALIZED_METADATA_IDX), target_platform=_get_str(engine_info, TARGET_PLATFORM_IDX), + # Content-keyed, so identical engines in different methods share one + # entry and different engines can never collide on a key. + engine_key="tensorrt_engine_" + hashlib.sha256(engine_bytes).hexdigest(), + ) + # The engine goes to named data by reference and is streamed into the .pte, + # so the delegate blob holds only the header and metadata. + named_data = NamedDataStore() + named_data.add_named_data(metadata.engine_key, engine_bytes, alignment=16) + return PreprocessResult( + processed_bytes=serialize_engine(b"", metadata), + data_store_output=named_data.get_named_data_store_output(), ) - blob = serialize_engine(engine_info[ENGINE_IDX], metadata) - return PreprocessResult(processed_bytes=blob) diff --git a/py/torch_tensorrt/executorch/serialization.py b/py/torch_tensorrt/executorch/serialization.py index a8efab0ec4d..9b8050b5663 100644 --- a/py/torch_tensorrt/executorch/serialization.py +++ b/py/torch_tensorrt/executorch/serialization.py @@ -51,6 +51,9 @@ class TensorRTBlobMetadata: device_id: int = 0 serialized_metadata: str = "" target_platform: str = "" + # Named-data key of the engine when it is stored outside the blob, which then + # carries an empty engine. Empty means the engine follows the metadata inline. + engine_key: str = "" def to_json(self) -> bytes: # Keep field order stable because the C++ parser is intentionally small. @@ -80,6 +83,8 @@ def to_json(self) -> bytes: "serialized_metadata": self.serialized_metadata, "target_platform": self.target_platform, } + if self.engine_key: + data["engine_key"] = self.engine_key return json.dumps(data, separators=(",", ":")).encode("utf-8") @classmethod @@ -103,6 +108,7 @@ def from_json(cls, data: bytes) -> "TensorRTBlobMetadata": device_id=parsed.get("device_id", 0), serialized_metadata=parsed.get("serialized_metadata", ""), target_platform=parsed.get("target_platform", ""), + engine_key=parsed.get("engine_key", ""), ) diff --git a/tests/cpp/executorch/test_executorch_blob_header.cpp b/tests/cpp/executorch/test_executorch_blob_header.cpp index d26bda2b448..1d7d6607a43 100644 --- a/tests/cpp/executorch/test_executorch_blob_header.cpp +++ b/tests/cpp/executorch/test_executorch_blob_header.cpp @@ -759,6 +759,23 @@ TEST(ExecuTorchTensorRTBlobHeader, RejectsUnknownFutureMagic) { EXPECT_FALSE(TensorRTBlobHeader::parse(blob.data(), blob.size(), header)); } +TEST(ExecuTorchTensorRTBlobHeader, ReadsTheEngineKeyOfAnEngineStoredAsNamedData) { + const std::string bindings = R"({"io_bindings":[{"name":"x","is_input":true}],)"; + const auto blob = make_blob(bindings + R"("engine_key":"tensorrt_engine_ab"})", 0); + TensorRTBlobHeader header; + ASSERT_TRUE(TensorRTBlobHeader::parse(blob.data(), blob.size(), header)); + EXPECT_EQ(header.engine_key, "tensorrt_engine_ab"); + EXPECT_EQ(header.engine_size, 0u); + + // Inline and named at once leaves one of the two engines unread, and neither leaves none to load. + const auto both = make_blob(bindings + R"("engine_key":"tensorrt_engine_ab"})", 4); + EXPECT_FALSE(TensorRTBlobHeader::parse(both.data(), both.size(), header)); + const auto neither = make_blob(VALID_METADATA, 0); + EXPECT_FALSE(TensorRTBlobHeader::parse(neither.data(), neither.size(), header)); + const auto empty_key = make_blob(bindings + R"("engine_key":""})", 0); + EXPECT_FALSE(TensorRTBlobHeader::parse(empty_key.data(), empty_key.size(), header)); +} + } // namespace } // namespace executorch_backend } // namespace torch_tensorrt diff --git a/tests/py/dynamo/executorch/test_backend.py b/tests/py/dynamo/executorch/test_backend.py index 1cd81964937..4dc6e434ec8 100644 --- a/tests/py/dynamo/executorch/test_backend.py +++ b/tests/py/dynamo/executorch/test_backend.py @@ -3,9 +3,9 @@ import ast import operator -import struct from pathlib import Path from types import SimpleNamespace +from typing import Any import pytest @@ -28,8 +28,6 @@ _get_engine_info_from_edge_program, ) from torch_tensorrt.executorch.serialization import ( # noqa: E402 - HEADER_FORMAT, - HEADER_SIZE, TENSORRT_MAGIC, deserialize_engine, ) @@ -131,6 +129,14 @@ def _engine_tensor(payload: bytes) -> torch.Tensor: return torch.frombuffer(bytearray(payload), dtype=torch.uint8) +def _named_engine(result: Any) -> bytes: + """The engine a preprocess result stores as named data; its blob holds none.""" + blob_engine, metadata = deserialize_engine(result.processed_bytes) + assert blob_engine == b"" + store = result.data_store_output + return store.buffers[store.pte_data[metadata.engine_key].buffer_index] + + @pytest.mark.unit def test_no_op_placeholder_schema_matches_serialized_engine_layout(): meta_ops_path = ( @@ -191,8 +197,8 @@ def test_preprocess_serializes_engine_blob(): assert isinstance(result.processed_bytes, bytes) assert result.processed_bytes[:4] == TENSORRT_MAGIC - engine, metadata = deserialize_engine(result.processed_bytes) - assert engine == b"engine-bytes" + assert _named_engine(result) == b"engine-bytes" + _, metadata = deserialize_engine(result.processed_bytes) assert metadata.device_id == 2 assert [binding.name for binding in metadata.io_bindings] == ["x", "y"] assert [binding.is_input for binding in metadata.io_bindings] == [True, False] @@ -202,9 +208,6 @@ def test_preprocess_serializes_engine_blob(): def test_preprocess_serializes_only_the_engine_tensors_extent(): # A tensor that views part of a larger buffer. Serializing the whole storage # instead of the tensor's own extent pads the engine with the trailing bytes. - # The recorded engine size is what exposes it: deserialize_engine trims the - # blob back to that size, so an over-long engine still round-trips and only - # the size gives it away. payload = b"engine-bytes" backing = torch.frombuffer(bytearray(payload + b"TRAILING"), dtype=torch.uint8) engine_info = [""] * SERIALIZATION_LEN @@ -216,12 +219,7 @@ def test_preprocess_serializes_only_the_engine_tensors_extent(): result = TensorRTBackend.preprocess(edge_program, []) - _, _, _, _, engine_size, _ = struct.unpack( - HEADER_FORMAT, result.processed_bytes[:HEADER_SIZE] - ) - assert engine_size == len(payload) - engine, _ = deserialize_engine(result.processed_bytes) - assert engine == payload + assert _named_engine(result) == payload @pytest.mark.unit @@ -246,36 +244,23 @@ def forward(self, x): @pytest.mark.unit -def test_preprocess_hands_the_engine_to_the_blob_without_copying_it(monkeypatch): - # The blob concatenation is the one copy of the engine preprocess needs. Turning - # the engine tensor into bytes first would hold one more copy at the same time, - # and an engine can be several gigabytes. - import numpy as np - - from torch_tensorrt.executorch import backend as backend_module +def test_preprocess_hands_the_staged_engine_bytes_to_named_data_without_copying(): + # An engine can be several gigabytes. The export stages it as a view of the + # module's own bytes, and the .pte is written from named data by reference, so + # the same object must come out the other end -- any copy is a whole engine more. + from torch_tensorrt.executorch._export_utils import engine_bytes_tensor - engine_tensor = _engine_tensor(b"engine-bytes") + payload = b"engine-bytes" engine_info = [""] * SERIALIZATION_LEN - engine_info[ENGINE_IDX] = engine_tensor + engine_info[ENGINE_IDX] = engine_bytes_tensor(payload) engine_info[DEVICE_IDX] = "0%8%0%0%GPU" engine_info[INPUT_BINDING_NAMES_IDX] = "x" engine_info[OUTPUT_BINDING_NAMES_IDX] = "y" edge_program = _build_edge_program(engine_info) - received = [] - real_serialize_engine = backend_module.serialize_engine - - def recording_serialize_engine(engine_bytes, metadata): - received.append(engine_bytes) - return real_serialize_engine(engine_bytes, metadata) - - monkeypatch.setattr(backend_module, "serialize_engine", recording_serialize_engine) result = TensorRTBackend.preprocess(edge_program, []) - assert len(received) == 1 - assert np.shares_memory(np.asarray(received[0]), engine_tensor.numpy()) - engine, _ = deserialize_engine(result.processed_bytes) - assert engine == b"engine-bytes" + assert _named_engine(result) is payload @pytest.mark.unit diff --git a/tests/py/dynamo/executorch/test_engine_info_accessors.py b/tests/py/dynamo/executorch/test_engine_info_accessors.py index 98ecf8b14c6..6b77f597993 100644 --- a/tests/py/dynamo/executorch/test_engine_info_accessors.py +++ b/tests/py/dynamo/executorch/test_engine_info_accessors.py @@ -74,6 +74,13 @@ def __getstate__(self): return (_record(base64.b64encode(ENGINE_BYTES).decode()),) +class RuntimeWithTensorAccessor(RuntimeWithAccessor): + """Engine from a runtime that serializes into a buffer it owns, like the C++ one.""" + + def serialized_engine_tensor(self): + return torch.frombuffer(bytearray(ENGINE_BYTES), dtype=torch.uint8) + + class _Program: """Stand-in for ExportedProgram carrying what the two passes touch.""" @@ -185,3 +192,26 @@ def test_metadata_only_record_is_not_cross_served_as_engine_bytes(): "metadata-only record cached by validation rather than re-resolving them" ) assert engine.getstate_calls == 1, "the engine record still must be read once" + + +@pytest.mark.unit +def test_rewrite_stages_the_engine_once_as_bytes(): + """The buffer the rewrite registers lives until the save ends, and ExecuTorch + writes named data only from ``bytes``. A buffer still viewing the runtime's + serialization makes preprocess copy it, holding every engine twice.""" + import numpy as np + + program, _ = _program_with_engine(RuntimeWithTensorAccessor()) + _export_utils.replace_execute_engine(program, {}) + + no_op = torch.ops.tensorrt.no_op_placeholder_for_execute_engine.default + (no_op_node,) = [n for n in program.graph_module.graph.nodes if n.target is no_op] + engine_buffer = getattr( + program.graph_module, no_op_node.args[1 + ENGINE_IDX].target + ) + + engine_bytes = engine_buffer._trt_engine_bytes + assert engine_bytes == ENGINE_BYTES + assert np.shares_memory( + engine_buffer.numpy(), np.frombuffer(engine_bytes, dtype=np.uint8) + ), "the registered buffer must view the bytes preprocess hands to named data"