TensorRT 11.4.0
NvInferRuntime.h
Go to the documentation of this file.
1/*
2 * SPDX-FileCopyrightText: Copyright (c) 1993-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
3 * SPDX-License-Identifier: Apache-2.0
4 *
5 * Licensed under the Apache License, Version 2.0 (the "License");
6 * you may not use this file except in compliance with the License.
7 * You may obtain a copy of the License at
8 *
9 * http://www.apache.org/licenses/LICENSE-2.0
10 *
11 * Unless required by applicable law or agreed to in writing, software
12 * distributed under the License is distributed on an "AS IS" BASIS,
13 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14 * See the License for the specific language governing permissions and
15 * limitations under the License.
16 */
17
18#ifndef NV_INFER_RUNTIME_H
19#define NV_INFER_RUNTIME_H
20
26
27#include "NvInferImpl.h" // IWYU pragma: export
28#define NV_INFER_INTERNAL_INCLUDE 1
29#include "NvInferPluginBase.h" // IWYU pragma: export
30#undef NV_INFER_INTERNAL_INCLUDE
31#include "NvInferRuntimeCommon.h" // IWYU pragma: export
32
33namespace nvinfer1
34{
35
36class IExecutionContext;
37class ICudaEngine;
38class IPluginFactory;
39class IEngineInspector;
40
49
51{
52protected:
53 INoCopy() = default;
54 virtual ~INoCopy() = default;
55 INoCopy(INoCopy const& other) = delete;
56 INoCopy& operator=(INoCopy const& other) = delete;
57 INoCopy(INoCopy&& other) = delete;
58 INoCopy& operator=(INoCopy&& other) = delete;
59};
60
75enum class EngineCapability : int32_t
76{
81 kSTANDARD = 0,
82
89 kSAFETY = 1,
90
97};
98
100template <>
102{
103 static constexpr int32_t kVALUE = 3;
104};
105
132{
133public:
135 void const* values;
136 int64_t count;
137};
138
149class IHostMemory : public INoCopy
150{
151public:
152 virtual ~IHostMemory() noexcept = 0;
153
155 void* data() const noexcept
156 {
157 return mImpl->data();
158 }
159
161 std::size_t size() const noexcept
162 {
163 return mImpl->size();
164 }
165
167 DataType type() const noexcept
168 {
169 return mImpl->type();
170 }
171
172protected:
173 apiv::VHostMemory* mImpl;
174};
175
176inline IHostMemory::~IHostMemory() noexcept = default;
177
188enum class DimensionOperation : int32_t
189{
190 kSUM = 0,
191 kPROD = 1,
192 kMAX = 2,
193 kMIN = 3,
194 kSUB = 4,
195 kEQUAL = 5,
196 kLESS = 6,
197 kFLOOR_DIV = 7,
198 kCEIL_DIV = 8
199};
200
202template <>
204{
205 static constexpr int32_t kVALUE = 9;
206};
207
213enum class TensorLocation : int32_t
214{
215 kDEVICE = 0,
216 kHOST = 1,
217};
218
220template <>
222{
223 static constexpr int32_t kVALUE = 2;
224};
225
239{
240public:
244 bool isConstant() const noexcept
245 {
246 return mImpl->isConstant();
247 }
248
255 int64_t getConstantValue() const noexcept
256 {
257 return mImpl->getConstantValue();
258 }
259
260protected:
261 apiv::VDimensionExpr* mImpl;
262 virtual ~IDimensionExpr() noexcept = 0;
263
264public:
270 bool isSizeTensor() const noexcept
271 {
272 return mImpl->isSizeTensor();
273 }
274};
275
276inline IDimensionExpr::~IDimensionExpr() noexcept = default;
277
295class IExprBuilder : public INoCopy
296{
297public:
301 IDimensionExpr const* constant(int64_t value) noexcept
302 {
303 return mImpl->constant(value);
304 }
305
313 DimensionOperation op, IDimensionExpr const& first, IDimensionExpr const& second) noexcept
314 {
315 return mImpl->operation(op, first, second);
316 }
317
318protected:
319 apiv::VExprBuilder* mImpl;
320 virtual ~IExprBuilder() noexcept = 0;
321
322public:
347 IDimensionExpr const* declareSizeTensor(int32_t outputIndex, IDimensionExpr const& opt, IDimensionExpr const& upper)
348 {
349 return mImpl->declareSizeTensor(outputIndex, opt, upper);
350 }
351};
352
353inline IExprBuilder::~IExprBuilder() noexcept = default;
354
361{
362public:
363 int32_t nbDims;
365};
366
373{
376
379
382
385};
386
418{
419public:
420 IPluginV2DynamicExt* clone() const noexcept override = 0;
421
446 virtual DimsExprs getOutputDimensions(
447 int32_t outputIndex, DimsExprs const* inputs, int32_t nbInputs, IExprBuilder& exprBuilder) noexcept = 0;
448
452 static constexpr int32_t kFORMAT_COMBINATION_LIMIT = 100;
453
486 virtual bool supportsFormatCombination(
487 int32_t pos, PluginTensorDesc const* inOut, int32_t nbInputs, int32_t nbOutputs) noexcept = 0;
488
524 virtual void configurePlugin(DynamicPluginTensorDesc const* in, int32_t nbInputs,
525 DynamicPluginTensorDesc const* out, int32_t nbOutputs) noexcept = 0;
526
536 virtual size_t getWorkspaceSize(PluginTensorDesc const* inputs, int32_t nbInputs, PluginTensorDesc const* outputs,
537 int32_t nbOutputs) const noexcept = 0;
538
551 virtual int32_t enqueue(PluginTensorDesc const* inputDesc, PluginTensorDesc const* outputDesc,
552 void const* const* inputs, void* const* outputs, void* workspace, cudaStream_t stream) noexcept = 0;
553
554protected:
562 int32_t getTensorRTVersion() const noexcept override
563 {
564 return (static_cast<int32_t>(PluginVersion::kV2_DYNAMICEXT) << 24 | (NV_TENSORRT_VERSION & 0xFFFFFF));
565 }
566
567 virtual ~IPluginV2DynamicExt() noexcept {}
568
569private:
570 // Following are obsolete base class methods, and must not be implemented or used.
571
575 void configurePlugin(Dims const*, int32_t, Dims const*, int32_t, DataType const*, DataType const*, bool const*,
576 bool const*, PluginFormat, int32_t) noexcept final
577 {
578 }
579
583 bool supportsFormat(DataType, PluginFormat) const noexcept final
584 {
585 return false;
586 }
587
591 Dims getOutputDimensions(int32_t, Dims const*, int32_t) noexcept final
592 {
593 return Dims{-1, {}};
594 }
595
599 size_t getWorkspaceSize(int32_t) const noexcept final
600 {
601 return 0;
602 }
603
607 int32_t enqueue(int32_t, void const* const*, void* const*, void*, cudaStream_t) noexcept final
608 {
609 return 1;
610 }
611};
612
613namespace v_1_0
614{
619{
620public:
625 ~IStreamReader() override = default;
626 IStreamReader() = default;
627
631 InterfaceInfo getInterfaceInfo() const noexcept override
632 {
633 return InterfaceInfo{"IStreamReader", 1, 0};
634 }
635
644 virtual int64_t read(void* destination, int64_t nbBytes) = 0;
645
646protected:
647 IStreamReader(IStreamReader const&) = default;
651};
652
654{
655public:
660 ~IStreamWriter() override = default;
661 IStreamWriter() = default;
662
666 InterfaceInfo getInterfaceInfo() const noexcept override
667 {
668 return InterfaceInfo{"IStreamWriter", 1, 0};
669 }
670
680 virtual int64_t write(void const* data, int64_t nbBytes) = 0;
681
682protected:
683 IStreamWriter(IStreamWriter const&) = default;
687};
688} // namespace v_1_0
689
701
711
716enum class SeekPosition : int32_t
717{
719 kSET = 0,
720
722 kCUR = 1,
723
725 kEND = 2,
726};
727
728namespace v_1_0
729{
731{
732public:
737 ~IStreamReaderV2() override = default;
738 IStreamReaderV2() = default;
739
743 InterfaceInfo getInterfaceInfo() const noexcept override
744 {
745 return InterfaceInfo{"IStreamReaderV2", 1, 0};
746 }
747
758 virtual int64_t read(void* destination, int64_t nbBytes, cudaStream_t stream) noexcept = 0;
759
768 virtual bool seek(int64_t offset, SeekPosition where) noexcept = 0;
769
770protected:
775};
776} // namespace v_1_0
777
788
803{
804public:
809 virtual IGpuAllocator* getGpuAllocator() const noexcept = 0;
810
815 virtual IErrorRecorder* getErrorRecorder() const noexcept = 0;
816 virtual ~IPluginResourceContext() noexcept = 0;
817
818protected:
822 IPluginResourceContext& operator=(IPluginResourceContext const&) & = default;
824};
825
826inline IPluginResourceContext::~IPluginResourceContext() noexcept = default;
827
828namespace v_1_0
829{
831{
832public:
836 InterfaceInfo getInterfaceInfo() const noexcept override
837 {
838 return InterfaceInfo{"PLUGIN_V3ONE_CORE", 1, 0};
839 }
840
849 virtual AsciiChar const* getPluginName() const noexcept = 0;
850
859 virtual AsciiChar const* getPluginVersion() const noexcept = 0;
860
870 virtual AsciiChar const* getPluginNamespace() const noexcept = 0;
871};
872
874{
875public:
881 static constexpr int32_t kDEFAULT_FORMAT_COMBINATION_LIMIT = 100;
882
886 InterfaceInfo getInterfaceInfo() const noexcept override
887 {
888 return InterfaceInfo{"PLUGIN_V3ONE_BUILD", 1, 0};
889 }
890
910 virtual int32_t configurePlugin(DynamicPluginTensorDesc const* in, int32_t nbInputs,
911 DynamicPluginTensorDesc const* out, int32_t nbOutputs) noexcept = 0;
912
929 virtual int32_t getOutputDataTypes(
930 DataType* outputTypes, int32_t nbOutputs, DataType const* inputTypes, int32_t nbInputs) const noexcept = 0;
931
953 virtual int32_t getOutputShapes(DimsExprs const* inputs, int32_t nbInputs, DimsExprs const* shapeInputs,
954 int32_t nbShapeInputs, DimsExprs* outputs, int32_t nbOutputs, IExprBuilder& exprBuilder) noexcept = 0;
955
991 int32_t pos, DynamicPluginTensorDesc const* inOut, int32_t nbInputs, int32_t nbOutputs) noexcept = 0;
992
998 virtual int32_t getNbOutputs() const noexcept = 0;
999
1009 virtual size_t getWorkspaceSize(DynamicPluginTensorDesc const* /* inputs */, int32_t /* nbInputs */,
1010 DynamicPluginTensorDesc const* /* outputs */, int32_t /* nbOutputs */) const noexcept
1011 {
1012 return 0;
1013 }
1014
1046 virtual int32_t getValidTactics(int32_t* /* tactics */, int32_t /* nbTactics */) noexcept
1047 {
1048 return 0;
1049 }
1050
1054 virtual int32_t getNbTactics() noexcept
1055 {
1056 return 0;
1057 }
1058
1070 virtual char const* getTimingCacheID() noexcept
1071 {
1072 return nullptr;
1073 }
1074
1078 virtual int32_t getFormatCombinationLimit() noexcept
1079 {
1080 return kDEFAULT_FORMAT_COMBINATION_LIMIT;
1081 }
1082
1089 virtual char const* getMetadataString() noexcept
1090 {
1091 return nullptr;
1092 }
1093};
1094
1096{
1097public:
1101 InterfaceInfo getInterfaceInfo() const noexcept override
1102 {
1103 return InterfaceInfo{"PLUGIN_V3ONE_RUNTIME", 1, 0};
1104 }
1105
1113 virtual int32_t setTactic(int32_t /* tactic */) noexcept
1114 {
1115 return 0;
1116 }
1117
1136 virtual int32_t onShapeChange(
1137 PluginTensorDesc const* in, int32_t nbInputs, PluginTensorDesc const* out, int32_t nbOutputs) noexcept = 0;
1138
1152 virtual int32_t enqueue(PluginTensorDesc const* inputDesc, PluginTensorDesc const* outputDesc,
1153 void const* const* inputs, void* const* outputs, void* workspace, cudaStream_t stream) noexcept = 0;
1154
1174 virtual IPluginV3* attachToContext(IPluginResourceContext* context) noexcept = 0;
1175
1181
1185 virtual PluginFieldCollection const* getFieldsToSerialize() noexcept = 0;
1186};
1187} // namespace v_1_0
1188
1189namespace v_2_0
1190{
1191
1193{
1194public:
1195 InterfaceInfo getInterfaceInfo() const noexcept override
1196 {
1197 return InterfaceInfo{"PLUGIN_V3ONE_BUILD", 2, 0};
1198 }
1199
1229 virtual int32_t getAliasedInput(int32_t /* outputIndex */) noexcept
1230 {
1231 return -1;
1232 }
1233};
1234
1235} // namespace v_2_0
1236
1247
1259
1271
1280
1283namespace v_1_0
1284{
1286{
1287public:
1291 InterfaceInfo getInterfaceInfo() const noexcept override
1292 {
1293 return {"IProfiler", 1, 0};
1294 }
1295
1303 virtual void reportLayerTime(char const* layerName, float ms) noexcept = 0;
1304
1305 ~IProfiler() override = default;
1306};
1307} // namespace v_1_0
1308
1321
1329enum class WeightsRole : int32_t
1330{
1331 kKERNEL = 0,
1332 kBIAS = 1,
1333 kSHIFT = 2,
1334 kSCALE = 3,
1335 kCONSTANT = 4,
1336 kANY = 5,
1337};
1338
1340template <>
1342{
1343 static constexpr int32_t kVALUE = 6;
1344};
1345
1351enum class DeviceType : int32_t
1352{
1353 kGPU = 0,
1354 kDLA = 1,
1355};
1356
1358template <>
1360{
1361 static constexpr int32_t kVALUE = 2;
1362};
1363
1374enum class TempfileControlFlag : int32_t
1375{
1378
1383};
1384
1386template <>
1388{
1389 static constexpr int32_t kVALUE = 2;
1390};
1391
1398using TempfileControlFlags = uint32_t;
1399
1431enum class TensorFormat : int32_t
1432{
1439 kLINEAR = 0,
1440
1445 kCHW2 = 1,
1446
1450 kHWC8 = 2,
1451
1465 kCHW4 = 3,
1466
1473 kCHW16 = 4,
1474
1482 kCHW32 = 5,
1483
1488 kDHWC8 = 6,
1489
1494 kCDHW32 = 7,
1495
1499 kHWC = 8,
1500
1509 kDLA_LINEAR = 9,
1510
1524 kDLA_HWC4 = 10,
1525
1530 kHWC16 = 11,
1531
1536 kDHWC = 12
1537};
1538
1540template <>
1542{
1544 static constexpr int32_t kVALUE = 13;
1545};
1546
1552enum class AllocatorFlag : int32_t
1553{
1555 kRESIZABLE = 0,
1556};
1557
1559template <>
1561{
1563 static constexpr int32_t kVALUE = 1;
1564};
1565
1566using AllocatorFlags = uint32_t;
1567
1576{
1578 kDEFAULT = 0,
1585 kSHARED_STATIC = 1,
1586};
1587
1589template <>
1591{
1593 static constexpr int32_t kVALUE = 2;
1594};
1595
1598namespace v_1_0
1599{
1600
1614{
1615public:
1619 InterfaceInfo getInterfaceInfo() const noexcept override
1620 {
1621 return {"ILogger", 1, 0};
1622 }
1623
1629 enum class Severity : int32_t
1630 {
1632 kINTERNAL_ERROR = 0,
1634 kERROR = 1,
1636 kWARNING = 2,
1638 kINFO = 3,
1640 kVERBOSE = 4,
1641 };
1642
1661 virtual void log(Severity severity, AsciiChar const* msg) noexcept = 0;
1662
1663 ILogger() = default;
1664 ~ILogger() override = default;
1665
1666protected:
1667 // @cond SuppressDoxyWarnings
1668 ILogger(ILogger const&) = default;
1669 ILogger(ILogger&&) = default;
1670 ILogger& operator=(ILogger const&) & = default;
1671 ILogger& operator=(ILogger&&) & = default;
1672 // @endcond
1673};
1674
1675} // namespace v_1_0
1676
1677using ILogger = v_1_0::ILogger;
1678
1680template <>
1682{
1684 static constexpr int32_t kVALUE = 5;
1685};
1686
1687namespace v_1_0
1688{
1689
1691{
1692public:
1718 uint64_t const size, uint64_t const alignment, AllocatorFlags const flags) noexcept = 0;
1719
1720 ~IGpuAllocator() override = default;
1721 IGpuAllocator() = default;
1722
1760 virtual void* reallocate(void* const /*baseAddr*/, uint64_t /*alignment*/, uint64_t /*newSize*/) noexcept
1761 {
1762 return nullptr;
1763 }
1764
1783 TRT_DEPRECATED virtual bool deallocate(void* const memory) noexcept = 0;
1784
1813 virtual void* allocateAsync(
1814 uint64_t const size, uint64_t const alignment, AllocatorFlags const flags, cudaStream_t /*stream*/) noexcept
1815 {
1816 return allocate(size, alignment, flags);
1817 }
1846 virtual bool deallocateAsync(void* const memory, cudaStream_t /*stream*/) noexcept
1847 {
1848 return deallocate(memory);
1849 }
1850
1854 InterfaceInfo getInterfaceInfo() const noexcept override
1855 {
1856 return {"IGpuAllocator", 1, 0};
1857 }
1858
1859protected:
1860 // @cond SuppressDoxyWarnings
1861 IGpuAllocator(IGpuAllocator const&) = default;
1862 IGpuAllocator(IGpuAllocator&&) = default;
1863 IGpuAllocator& operator=(IGpuAllocator const&) & = default;
1864 IGpuAllocator& operator=(IGpuAllocator&&) & = default;
1865 // @endcond
1866};
1867
1868} // namespace v_1_0
1869
1891
1892
1900class IRuntime : public INoCopy
1901{
1902public:
1903 virtual ~IRuntime() noexcept = 0;
1904
1919 void setDLACore(int32_t dlaCore) noexcept
1920 {
1921 mImpl->setDLACore(dlaCore);
1922 }
1923
1929 int32_t getDLACore() const noexcept
1930 {
1931 return mImpl->getDLACore();
1932 }
1933
1937 int32_t getNbDLACores() const noexcept
1938 {
1939 return mImpl->getNbDLACores();
1940 }
1941
1955 {
1956 return mImpl->setDLAWorkspaceAllocationStrategy(strategy);
1957 }
1958
1965 {
1966 return mImpl->getDLAWorkspaceAllocationStrategy();
1967 }
1968
1980 void setGpuAllocator(IGpuAllocator* allocator) noexcept
1981 {
1982 mImpl->setGpuAllocator(allocator);
1983 }
1984
1996 //
1999 void setErrorRecorder(IErrorRecorder* recorder) noexcept
2000 {
2001 mImpl->setErrorRecorder(recorder);
2002 }
2003
2015 {
2016 return mImpl->getErrorRecorder();
2017 }
2018
2032 ICudaEngine* deserializeCudaEngine(void const* blob, std::size_t size) noexcept
2033 {
2034 return mImpl->deserializeCudaEngine(blob, size);
2035 }
2036
2056 {
2057 return mImpl->deserializeCudaEngineV2(streamReader);
2058 }
2059
2065 ILogger* getLogger() const noexcept
2066 {
2067 return mImpl->getLogger();
2068 }
2069
2080 bool setMaxThreads(int32_t maxThreads) noexcept
2081 {
2082 return mImpl->setMaxThreads(maxThreads);
2083 }
2084
2094 int32_t getMaxThreads() const noexcept
2095 {
2096 return mImpl->getMaxThreads();
2097 }
2098
2129 void setTemporaryDirectory(char const* path) noexcept
2130 {
2131 return mImpl->setTemporaryDirectory(path);
2132 }
2133
2140 char const* getTemporaryDirectory() const noexcept
2141 {
2142 return mImpl->getTemporaryDirectory();
2143 }
2144
2157 {
2158 return mImpl->setTempfileControlFlags(flags);
2159 }
2160
2169 {
2170 return mImpl->getTempfileControlFlags();
2171 }
2172
2179 {
2180 return mImpl->getPluginRegistry();
2181 }
2182
2196 IRuntime* loadRuntime(char const* path) noexcept
2197 {
2198 return mImpl->loadRuntime(path);
2199 }
2200
2208 void setEngineHostCodeAllowed(bool allowed) noexcept
2209 {
2210 return mImpl->setEngineHostCodeAllowed(allowed);
2211 }
2212
2218 bool getEngineHostCodeAllowed() const noexcept
2219 {
2220 return mImpl->getEngineHostCodeAllowed();
2221 }
2222
2223
2224protected:
2225 apiv::VRuntime* mImpl{};
2226};
2227
2228inline IRuntime::~IRuntime() noexcept = default;
2229
2237class IRefitter : public INoCopy
2238{
2239public:
2240 virtual ~IRefitter() noexcept = 0;
2241
2257 bool setWeights(char const* layerName, WeightsRole role, Weights weights) noexcept
2258 {
2259 return mImpl->setWeights(layerName, role, weights);
2260 }
2261
2274 bool refitCudaEngine() noexcept
2275 {
2276 return mImpl->refitCudaEngine();
2277 }
2278
2295 int32_t getMissing(int32_t size, char const** layerNames, WeightsRole* roles) noexcept
2296 {
2297 return mImpl->getMissing(size, layerNames, roles);
2298 }
2299
2312 int32_t getAll(int32_t size, char const** layerNames, WeightsRole* roles) noexcept
2313 {
2314 return mImpl->getAll(size, layerNames, roles);
2315 }
2316
2328 //
2331 void setErrorRecorder(IErrorRecorder* recorder) noexcept
2332 {
2333 mImpl->setErrorRecorder(recorder);
2334 }
2335
2347 {
2348 return mImpl->getErrorRecorder();
2349 }
2350
2371 bool setNamedWeights(char const* name, Weights weights) noexcept
2372 {
2373 return mImpl->setNamedWeights(name, weights);
2374 }
2375
2391 int32_t getMissingWeights(int32_t size, char const** weightsNames) noexcept
2392 {
2393 return mImpl->getMissingWeights(size, weightsNames);
2394 }
2395
2407 int32_t getAllWeights(int32_t size, char const** weightsNames) noexcept
2408 {
2409 return mImpl->getAllWeights(size, weightsNames);
2410 }
2411
2417 ILogger* getLogger() const noexcept
2418 {
2419 return mImpl->getLogger();
2420 }
2421
2433 bool setMaxThreads(int32_t maxThreads) noexcept
2434 {
2435 return mImpl->setMaxThreads(maxThreads);
2436 }
2437
2447 int32_t getMaxThreads() const noexcept
2448 {
2449 return mImpl->getMaxThreads();
2450 }
2451
2474 bool setNamedWeights(char const* name, Weights weights, TensorLocation location) noexcept
2475 {
2476 return mImpl->setNamedWeightsWithLocation(name, weights, location);
2477 }
2478
2490 Weights getNamedWeights(char const* weightsName) const noexcept
2491 {
2492 return mImpl->getNamedWeights(weightsName);
2493 }
2494
2506 TensorLocation getWeightsLocation(char const* weightsName) const noexcept
2507 {
2508 return mImpl->getWeightsLocation(weightsName);
2509 }
2510
2522 bool unsetNamedWeights(char const* weightsName) noexcept
2523 {
2524 return mImpl->unsetNamedWeights(weightsName);
2525 }
2526
2538 void setWeightsValidation(bool weightsValidation) noexcept
2539 {
2540 return mImpl->setWeightsValidation(weightsValidation);
2541 }
2542
2546 bool getWeightsValidation() const noexcept
2547 {
2548 return mImpl->getWeightsValidation();
2549 }
2550
2577 bool refitCudaEngineAsync(cudaStream_t stream) noexcept
2578 {
2579 return mImpl->refitCudaEngineAsync(stream);
2580 }
2581
2595 Weights getWeightsPrototype(char const* weightsName) const noexcept
2596 {
2597 return mImpl->getWeightsPrototype(weightsName);
2598 }
2599
2610 {
2611 return mImpl->releaseRefitResources();
2612 }
2613
2614protected:
2615 apiv::VRefitter* mImpl;
2616};
2617
2618inline IRefitter::~IRefitter() noexcept = default;
2619
2630enum class OptProfileSelector : int32_t
2631{
2632 kMIN = 0,
2633 kOPT = 1,
2634 kMAX = 2
2635};
2636
2638template <>
2640{
2641 static constexpr int32_t kVALUE = 3;
2642};
2643
2667{
2668public:
2696 bool setDimensions(char const* inputName, OptProfileSelector select, Dims const& dims) noexcept
2697 {
2698 return mImpl->setDimensions(inputName, select, dims);
2699 }
2700
2708 Dims getDimensions(char const* inputName, OptProfileSelector select) const noexcept
2709 {
2710 return mImpl->getDimensions(inputName, select);
2711 }
2712
2721 int32_t getNbShapeValues(char const* inputName) const noexcept
2722 {
2723 return mImpl->getNbShapeValues(inputName);
2724 }
2725
2739 bool setExtraMemoryTarget(float target) noexcept
2740 {
2741 return mImpl->setExtraMemoryTarget(target);
2742 }
2743
2751 float getExtraMemoryTarget() const noexcept
2752 {
2753 return mImpl->getExtraMemoryTarget();
2754 }
2755
2768 bool isValid() const noexcept
2769 {
2770 return mImpl->isValid();
2771 }
2772
2816 char const* inputName, OptProfileSelector select, int64_t const* values, int32_t nbValues) noexcept
2817 {
2818 return mImpl->setShapeValuesV2(inputName, select, values, nbValues);
2819 }
2820
2828 int64_t const* getShapeValuesV2(char const* inputName, OptProfileSelector select) const noexcept
2829 {
2830 return mImpl->getShapeValuesV2(inputName, select);
2831 }
2832
2854 void setProfileStream(cudaStream_t stream) noexcept
2855 {
2856 mImpl->setProfileStream(stream);
2857 }
2858
2866 cudaStream_t getProfileStream() const noexcept
2867 {
2868 return mImpl->getProfileStream();
2869 }
2870
2871protected:
2872 apiv::VOptimizationProfile* mImpl;
2873 virtual ~IOptimizationProfile() noexcept = 0;
2874};
2875
2876inline IOptimizationProfile::~IOptimizationProfile() noexcept = default;
2877
2885enum class TacticSource : int32_t
2886{
2891
2895};
2896
2898template <>
2900{
2901 static constexpr int32_t kVALUE = 2;
2902};
2903
2910using TacticSources = uint32_t;
2911
2921enum class ProfilingVerbosity : int32_t
2922{
2923 kLAYER_NAMES_ONLY = 0,
2924 kNONE = 1,
2925 kDETAILED = 2,
2926};
2927
2929template <>
2931{
2932 static constexpr int32_t kVALUE = 3;
2933};
2934
2941using SerializationFlags = uint32_t;
2942
2950enum class SerializationFlag : int32_t
2951{
2952 kEXCLUDE_WEIGHTS = 0,
2954 kINCLUDE_REFIT = 2,
2955};
2956
2958template <>
2960{
2961 static constexpr int32_t kVALUE = 3;
2962};
2963
2972{
2973public:
2974 virtual ~ISerializationConfig() noexcept = 0;
2975
2987 bool setFlags(SerializationFlags serializationFlags) noexcept
2988 {
2989 return mImpl->setFlags(serializationFlags);
2990 }
2991
3000 {
3001 return mImpl->getFlags();
3002 }
3003
3011 bool clearFlag(SerializationFlag serializationFlag) noexcept
3012 {
3013 return mImpl->clearFlag(serializationFlag);
3014 }
3015
3023 bool setFlag(SerializationFlag serializationFlag) noexcept
3024 {
3025 return mImpl->setFlag(serializationFlag);
3026 }
3027
3035 bool getFlag(SerializationFlag serializationFlag) const noexcept
3036 {
3037 return mImpl->getFlag(serializationFlag);
3038 }
3039
3040protected:
3041 apiv::VSerializationConfig* mImpl;
3042};
3043
3044inline ISerializationConfig::~ISerializationConfig() noexcept = default;
3045
3058{
3059 kSTATIC = 0,
3060 kON_PROFILE_CHANGE = 1,
3061 kUSER_MANAGED = 2,
3062};
3063
3066template <>
3068{
3069 static constexpr int32_t kVALUE = 3;
3070};
3071
3072
3080{
3081public:
3082 virtual ~IRuntimeConfig() noexcept = 0;
3083
3089 void setExecutionContextAllocationStrategy(ExecutionContextAllocationStrategy strategy) noexcept
3090 {
3091 return mImpl->setExecutionContextAllocationStrategy(strategy);
3092 }
3093
3100 {
3101 return mImpl->getExecutionContextAllocationStrategy();
3102 }
3103
3104
3105protected:
3106 apiv::VRuntimeConfig* mImpl;
3107}; // class IRuntimeConfig
3108
3109inline IRuntimeConfig::~IRuntimeConfig() noexcept = default;
3110
3111
3120enum class EngineStat : int32_t
3121{
3124
3127};
3128
3130template <>
3132{
3133 static constexpr int32_t kVALUE = 2;
3134};
3135
3143class ICudaEngine : public INoCopy
3144{
3145public:
3146 virtual ~ICudaEngine() noexcept = 0;
3147
3158 Dims getTensorShape(char const* tensorName) const noexcept
3159 {
3160 return mImpl->getTensorShape(tensorName);
3161 }
3162
3173 DataType getTensorDataType(char const* tensorName) const noexcept
3174 {
3175 return mImpl->getTensorDataType(tensorName);
3176 }
3177
3187 int32_t getNbLayers() const noexcept
3188 {
3189 return mImpl->getNbLayers();
3190 }
3191
3201 IHostMemory* serialize() const noexcept
3202 {
3203 return mImpl->serialize();
3204 }
3205
3220 {
3221 return mImpl->createExecutionContext(strategy);
3222 }
3223
3236 TensorLocation getTensorLocation(char const* tensorName) const noexcept
3237 {
3238 return mImpl->getTensorLocation(tensorName);
3239 }
3240
3256 bool isShapeInferenceIO(char const* tensorName) const noexcept
3257 {
3258 return mImpl->isShapeInferenceIO(tensorName);
3259 }
3260
3270 TensorIOMode getTensorIOMode(char const* tensorName) const noexcept
3271 {
3272 return mImpl->getTensorIOMode(tensorName);
3273 }
3274
3289 TRT_NODISCARD char const* getAliasedInputTensor(char const* tensorName) const noexcept
3290 {
3291 return mImpl->getAliasedInputTensor(tensorName);
3292 }
3293
3302 {
3303 return mImpl->createExecutionContextWithRuntimeConfig(runtimeConfig);
3304 }
3305
3315 {
3316 return mImpl->createRuntimeConfig();
3317 }
3318
3319
3329 int64_t getDeviceMemorySizeV2() const noexcept
3330 {
3331 return mImpl->getDeviceMemorySizeV2();
3332 }
3333
3343 int64_t getDeviceMemorySizeForProfileV2(int32_t profileIndex) const noexcept
3344 {
3345 return mImpl->getDeviceMemorySizeForProfileV2(profileIndex);
3346 }
3347
3353 bool isRefittable() const noexcept
3354 {
3355 return mImpl->isRefittable();
3356 }
3357
3374 int32_t getTensorBytesPerComponent(char const* tensorName) const noexcept
3375 {
3376 return mImpl->getTensorBytesPerComponent(tensorName);
3377 }
3378
3392 int32_t getTensorBytesPerComponent(char const* tensorName, int32_t profileIndex) const noexcept
3393 {
3394 return mImpl->getTensorBytesPerComponentV2(tensorName, profileIndex);
3395 }
3396
3413 int32_t getTensorComponentsPerElement(char const* tensorName) const noexcept
3414 {
3415 return mImpl->getTensorComponentsPerElement(tensorName);
3416 }
3417
3431 int32_t getTensorComponentsPerElement(char const* tensorName, int32_t profileIndex) const noexcept
3432 {
3433 return mImpl->getTensorComponentsPerElementV2(tensorName, profileIndex);
3434 }
3435
3446 TensorFormat getTensorFormat(char const* tensorName) const noexcept
3447 {
3448 return mImpl->getTensorFormat(tensorName);
3449 }
3450
3460 TensorFormat getTensorFormat(char const* tensorName, int32_t profileIndex) const noexcept
3461 {
3462 return mImpl->getTensorFormatV2(tensorName, profileIndex);
3463 }
3464
3484 char const* getTensorFormatDesc(char const* tensorName) const noexcept
3485 {
3486 return mImpl->getTensorFormatDesc(tensorName);
3487 }
3488
3507 char const* getTensorFormatDesc(char const* tensorName, int32_t profileIndex) const noexcept
3508 {
3509 return mImpl->getTensorFormatDescV2(tensorName, profileIndex);
3510 }
3511
3524 int32_t getTensorVectorizedDim(char const* tensorName) const noexcept
3525 {
3526 return mImpl->getTensorVectorizedDim(tensorName);
3527 }
3528
3540 int32_t getTensorVectorizedDim(char const* tensorName, int32_t profileIndex) const noexcept
3541 {
3542 return mImpl->getTensorVectorizedDimV2(tensorName, profileIndex);
3543 }
3544
3555 char const* getName() const noexcept
3556 {
3557 return mImpl->getName();
3558 }
3559
3566 int32_t getNbOptimizationProfiles() const noexcept
3567 {
3568 return mImpl->getNbOptimizationProfiles();
3569 }
3570
3586 Dims getProfileShape(char const* tensorName, int32_t profileIndex, OptProfileSelector select) const noexcept
3587 {
3588 return mImpl->getProfileShape(tensorName, profileIndex, select);
3589 }
3590
3602 {
3603 return mImpl->getEngineCapability();
3604 }
3605
3620 void setErrorRecorder(IErrorRecorder* recorder) noexcept
3621 {
3622 return mImpl->setErrorRecorder(recorder);
3623 }
3624
3636 {
3637 return mImpl->getErrorRecorder();
3638 }
3639
3652 {
3653 return mImpl->getTacticSources();
3654 }
3655
3664 {
3665 return mImpl->getProfilingVerbosity();
3666 }
3667
3674 {
3675 return mImpl->createEngineInspector();
3676 }
3677
3686 int32_t getNbIOTensors() const noexcept
3687 {
3688 return mImpl->getNbIOTensors();
3689 }
3690
3698 char const* getIOTensorName(int32_t index) const noexcept
3699 {
3700 return mImpl->getIOTensorName(index);
3701 }
3702
3710 {
3711 return mImpl->getHardwareCompatibilityLevel();
3712 }
3713
3724 int32_t getNbAuxStreams() const noexcept
3725 {
3726 return mImpl->getNbAuxStreams();
3727 }
3728
3735 {
3736 return mImpl->createSerializationConfig();
3737 }
3738
3755 {
3756 return mImpl->serializeWithConfig(config);
3757 }
3758
3770 int64_t getStreamableWeightsSize() const noexcept
3771 {
3772 return mImpl->getStreamableWeightsSize();
3773 }
3774
3809 bool setWeightStreamingBudgetV2(int64_t gpuMemoryBudget) noexcept
3810 {
3811 return mImpl->setWeightStreamingBudgetV2(gpuMemoryBudget);
3812 }
3813
3825 int64_t getWeightStreamingBudgetV2() const noexcept
3826 {
3827 return mImpl->getWeightStreamingBudgetV2();
3828 }
3829
3848 int64_t getWeightStreamingAutomaticBudget() const noexcept
3849 {
3850 return mImpl->getWeightStreamingAutomaticBudget();
3851 }
3852
3875 {
3876 return mImpl->getWeightStreamingScratchMemorySize();
3877 }
3878
3888 bool isDebugTensor(char const* name) const noexcept
3889 {
3890 return mImpl->isDebugTensor(name);
3891 }
3892
3911 char const* tensorName, int32_t profileIndex, OptProfileSelector select) const noexcept
3912 {
3913 return mImpl->getProfileTensorValuesV2(tensorName, profileIndex, select);
3914 }
3915
3938 int64_t getEngineStat(EngineStat stat) const noexcept
3939 {
3940 return mImpl->getEngineStat(stat);
3941 }
3942
3943
3944protected:
3945 apiv::VCudaEngine* mImpl;
3946};
3947
3948inline ICudaEngine::~ICudaEngine() noexcept = default;
3949
3950namespace v_1_0
3951{
3953{
3954public:
3958 InterfaceInfo getInterfaceInfo() const noexcept override
3959 {
3960 return {"IOutputAllocator", 1, 0};
3961 }
3962
3983 char const* /* tensorName */, void* /* currentMemory */, uint64_t /* size */, uint64_t /* alignment */) noexcept
3984 {
3985 return nullptr;
3986 }
3987
4011 [[maybe_unused]] char const* tensorName, [[maybe_unused]] void* currentMemory, [[maybe_unused]] uint64_t size,
4012 [[maybe_unused]] uint64_t alignment, cudaStream_t /* stream */)
4013 {
4014 return reallocateOutput(tensorName, currentMemory, size, alignment);
4015 }
4016
4025 virtual void notifyShape(char const* tensorName, Dims const& dims) noexcept = 0;
4026};
4027} // namespace v_1_0
4028
4037
4038namespace v_1_0
4039{
4041{
4042public:
4046 InterfaceInfo getInterfaceInfo() const noexcept override
4047 {
4048 return {"IDebugListener", 1, 0};
4049 }
4050
4064 virtual bool processDebugTensor(void const* addr, TensorLocation location, DataType type, Dims const& shape,
4065 char const* name, cudaStream_t stream)
4066 = 0;
4067
4068 ~IDebugListener() override = default;
4069};
4070} // namespace v_1_0
4071
4078
4090{
4091public:
4092 virtual ~IExecutionContext() noexcept = 0;
4093
4102 void setDebugSync(bool sync) noexcept
4103 {
4104 mImpl->setDebugSync(sync);
4105 }
4106
4112 bool getDebugSync() const noexcept
4113 {
4114 return mImpl->getDebugSync();
4115 }
4116
4122 void setProfiler(IProfiler* profiler) noexcept
4123 {
4124 mImpl->setProfiler(profiler);
4125 }
4126
4132 IProfiler* getProfiler() const noexcept
4133 {
4134 return mImpl->getProfiler();
4135 }
4136
4142 ICudaEngine const& getEngine() const noexcept
4143 {
4144 return mImpl->getEngine();
4145 }
4146
4156 void setName(char const* name) noexcept
4157 {
4158 mImpl->setName(name);
4159 }
4160
4166 char const* getName() const noexcept
4167 {
4168 return mImpl->getName();
4169 }
4170
4192 void setDeviceMemory(void* memory) noexcept
4193 {
4194 mImpl->setDeviceMemory(memory);
4195 }
4196
4213 void setDeviceMemoryV2(void* memory, int64_t size) noexcept
4214 {
4215 return mImpl->setDeviceMemoryV2(memory, size);
4216 }
4217
4234 Dims getTensorStrides(char const* tensorName) const noexcept
4235 {
4236 return mImpl->getTensorStrides(tensorName);
4237 }
4238
4239public:
4249 int32_t getOptimizationProfile() const noexcept
4250 {
4251 return mImpl->getOptimizationProfile();
4252 }
4253
4267 bool setInputShape(char const* tensorName, Dims const& dims) noexcept
4268 {
4269 return mImpl->setInputShape(tensorName, dims);
4270 }
4271
4304 Dims getTensorShape(char const* tensorName) const noexcept
4305 {
4306 return mImpl->getTensorShape(tensorName);
4307 }
4308
4320 bool allInputDimensionsSpecified() const noexcept
4321 {
4322 return mImpl->allInputDimensionsSpecified();
4323 }
4324
4339 void setErrorRecorder(IErrorRecorder* recorder) noexcept
4340 {
4341 mImpl->setErrorRecorder(recorder);
4342 }
4343
4355 {
4356 return mImpl->getErrorRecorder();
4357 }
4358
4371 bool executeV2(void* const* bindings) noexcept
4372 {
4373 return mImpl->executeV2(bindings);
4374 }
4375
4415 bool setOptimizationProfileAsync(int32_t profileIndex, cudaStream_t stream) noexcept
4416 {
4417 return mImpl->setOptimizationProfileAsync(profileIndex, stream);
4418 }
4419
4431 void setEnqueueEmitsProfile(bool enqueueEmitsProfile) noexcept
4432 {
4433 mImpl->setEnqueueEmitsProfile(enqueueEmitsProfile);
4434 }
4435
4443 bool getEnqueueEmitsProfile() const noexcept
4444 {
4445 return mImpl->getEnqueueEmitsProfile();
4446 }
4447
4473 bool reportToProfiler() const noexcept
4474 {
4475 return mImpl->reportToProfiler();
4476 }
4477
4517 bool setTensorAddress(char const* tensorName, void* data) noexcept
4518 {
4519 return mImpl->setTensorAddress(tensorName, data);
4520 }
4521
4534 void const* getTensorAddress(char const* tensorName) const noexcept
4535 {
4536 return mImpl->getTensorAddress(tensorName);
4537 }
4538
4557 bool setOutputTensorAddress(char const* tensorName, void* data) noexcept
4558 {
4559 return mImpl->setOutputTensorAddress(tensorName, data);
4560 }
4561
4579 bool setInputTensorAddress(char const* tensorName, void const* data) noexcept
4580 {
4581 return mImpl->setInputTensorAddress(tensorName, data);
4582 }
4583
4598 void* getOutputTensorAddress(char const* tensorName) const noexcept
4599 {
4600 return mImpl->getOutputTensorAddress(tensorName);
4601 }
4602
4631 int32_t inferShapes(int32_t nbMaxNames, char const** tensorNames) noexcept
4632 {
4633 return mImpl->inferShapes(nbMaxNames, tensorNames);
4634 }
4635
4649 {
4650 return mImpl->updateDeviceMemorySizeForShapes();
4651 }
4652
4664 bool setInputConsumedEvent(cudaEvent_t event) noexcept
4665 {
4666 return mImpl->setInputConsumedEvent(event);
4667 }
4668
4674 cudaEvent_t getInputConsumedEvent() const noexcept
4675 {
4676 return mImpl->getInputConsumedEvent();
4677 }
4678
4693 bool setOutputAllocator(char const* tensorName, IOutputAllocator* outputAllocator) noexcept
4694 {
4695 return mImpl->setOutputAllocator(tensorName, outputAllocator);
4696 }
4697
4706 IOutputAllocator* getOutputAllocator(char const* tensorName) const noexcept
4707 {
4708 return mImpl->getOutputAllocator(tensorName);
4709 }
4710
4724 int64_t getMaxOutputSize(char const* tensorName) const noexcept
4725 {
4726 return mImpl->getMaxOutputSize(tensorName);
4727 }
4728
4745 {
4746 return mImpl->setTemporaryStorageAllocator(allocator);
4747 }
4748
4755 {
4756 return mImpl->getTemporaryStorageAllocator();
4757 }
4758
4791 bool enqueueV3(cudaStream_t stream) noexcept
4792 {
4793 return mImpl->enqueueV3(stream);
4794 }
4795
4807 void setPersistentCacheLimit(size_t size) noexcept
4808 {
4809 mImpl->setPersistentCacheLimit(size);
4810 }
4811
4818 size_t getPersistentCacheLimit() const noexcept
4819 {
4820 return mImpl->getPersistentCacheLimit();
4821 }
4822
4842 bool setNvtxVerbosity(ProfilingVerbosity verbosity) noexcept
4843 {
4844 return mImpl->setNvtxVerbosity(verbosity);
4845 }
4846
4855 {
4856 return mImpl->getNvtxVerbosity();
4857 }
4858
4902 void setAuxStreams(cudaStream_t* auxStreams, int32_t nbStreams) noexcept
4903 {
4904 mImpl->setAuxStreams(auxStreams, nbStreams);
4905 }
4906
4914 bool setDebugListener(IDebugListener* listener) noexcept
4915 {
4916 return mImpl->setDebugListener(listener);
4917 }
4918
4925 {
4926 return mImpl->getDebugListener();
4927 }
4928
4943 bool setTensorDebugState(char const* name, bool flag) noexcept
4944 {
4945 return mImpl->setTensorDebugState(name, flag);
4946 }
4947
4955 bool getDebugState(char const* name) const noexcept
4956 {
4957 return mImpl->getDebugState(name);
4958 }
4959
4966 {
4967 return mImpl->getRuntimeConfig();
4968 }
4969
4978 bool setAllTensorsDebugState(bool flag) noexcept
4979 {
4980 return mImpl->setAllTensorsDebugState(flag);
4981 }
4982
4994 bool setUnfusedTensorsDebugState(bool flag) noexcept
4995 {
4996 return mImpl->setUnfusedTensorsDebugState(flag);
4997 }
4998
5004 bool getUnfusedTensorsDebugState() const noexcept
5005 {
5006 return mImpl->getUnfusedTensorsDebugState();
5007 }
5008
5022 bool setCommunicator(void* communicator) noexcept
5023 {
5024 return mImpl->setCommunicator(communicator);
5025 }
5026
5027protected:
5028 apiv::VExecutionContext* mImpl;
5029}; // class IExecutionContext
5030
5031inline IExecutionContext::~IExecutionContext() noexcept = default;
5032
5040enum class LayerInformationFormat : int32_t
5041{
5042 kONELINE = 0,
5043 kJSON = 1,
5044};
5045
5047template <>
5049{
5050 static constexpr int32_t kVALUE = 2;
5051};
5052
5069{
5070public:
5071 virtual ~IEngineInspector() noexcept = 0;
5072
5085 bool setExecutionContext(IExecutionContext const* context) noexcept
5086 {
5087 return mImpl->setExecutionContext(context);
5088 }
5089
5098 {
5099 return mImpl->getExecutionContext();
5100 }
5101
5122 char const* getLayerInformation(int32_t layerIndex, LayerInformationFormat format) const noexcept
5123 {
5124 return mImpl->getLayerInformation(layerIndex, format);
5125 }
5126
5145 char const* getEngineInformation(LayerInformationFormat format) const noexcept
5146 {
5147 return mImpl->getEngineInformation(format);
5148 }
5149
5164 void setErrorRecorder(IErrorRecorder* recorder) noexcept
5165 {
5166 mImpl->setErrorRecorder(recorder);
5167 }
5168
5180 {
5181 return mImpl->getErrorRecorder();
5182 }
5183
5184protected:
5185 apiv::VEngineInspector* mImpl;
5186}; // class IEngineInspector
5187
5188inline IEngineInspector::~IEngineInspector() noexcept = default;
5189
5190} // namespace nvinfer1
5191
5196extern "C" TENSORRTAPI void* createInferRuntime_INTERNAL(void* logger, int32_t version) noexcept;
5197
5202extern "C" TENSORRTAPI void* createInferRefitter_INTERNAL(void* engine, void* logger, int32_t version) noexcept;
5203
5207extern "C" TENSORRTAPI nvinfer1::IPluginRegistry* getPluginRegistry() noexcept;
5208
5214extern "C" TENSORRTAPI nvinfer1::ILogger* getLogger() noexcept;
5215
5216namespace nvinfer1
5217{
5218namespace // unnamed namespace avoids linkage surprises when linking objects built with different versions of this
5219 // header.
5220{
5226inline IRuntime* createInferRuntime(ILogger& logger) noexcept
5227{
5228 return static_cast<IRuntime*>(createInferRuntime_INTERNAL(&logger, NV_TENSORRT_VERSION));
5229}
5230
5237inline IRefitter* createInferRefitter(ICudaEngine& engine, ILogger& logger) noexcept
5238{
5239 return static_cast<IRefitter*>(createInferRefitter_INTERNAL(&engine, &logger, NV_TENSORRT_VERSION));
5240}
5241
5242} // namespace
5243
5255template <typename T>
5257{
5258public:
5260 {
5261 getPluginRegistry()->registerCreator(instance, "");
5262 }
5263
5264private:
5266 T instance{};
5267};
5268
5269} // namespace nvinfer1
5270
5271#define REGISTER_TENSORRT_PLUGIN(name) \
5272 static nvinfer1::PluginRegistrar<name> pluginRegistrar##name {}
5273
5274namespace nvinfer1
5275{
5278namespace v_1_0
5279{
5289{
5290public:
5294 InterfaceInfo getInterfaceInfo() const noexcept override
5295 {
5296 return {"ILoggerFinder", 1, 0};
5297 }
5298
5306 virtual ILogger* findLogger() = 0;
5307
5308protected:
5310 ~ILoggerFinder() override = default;
5311};
5312
5313} // namespace v_1_0
5314
5316
5319namespace v_1_0
5320{
5321
5323{
5324public:
5326 ~IGpuAsyncAllocator() override = default;
5327
5357 void* allocateAsync(uint64_t const size, uint64_t const alignment, AllocatorFlags const flags,
5358 cudaStream_t /*stream*/) noexcept override = 0;
5359
5385 bool deallocateAsync(void* const memory, cudaStream_t /*stream*/) noexcept override = 0;
5386
5411 uint64_t const size, uint64_t const alignment, AllocatorFlags const flags) noexcept override
5412 {
5413 return allocateAsync(size, alignment, flags, nullptr);
5414 }
5415
5434 TRT_DEPRECATED bool deallocate(void* const memory) noexcept override
5435 {
5436 return deallocateAsync(memory, nullptr);
5437 }
5438
5442 InterfaceInfo getInterfaceInfo() const noexcept override
5443 {
5444 return {"IGpuAllocator", 1, 0};
5445 }
5446};
5447
5449{
5450public:
5454 InterfaceInfo getInterfaceInfo() const noexcept override
5455 {
5456 return InterfaceInfo{"PLUGIN CREATOR_V3ONE", 1, 0};
5457 }
5458
5476 AsciiChar const* name, PluginFieldCollection const* fc, TensorRTPhase phase) noexcept = 0;
5477
5484 virtual PluginFieldCollection const* getFieldNames() noexcept = 0;
5485
5492 virtual AsciiChar const* getPluginName() const noexcept = 0;
5493
5500 virtual AsciiChar const* getPluginVersion() const noexcept = 0;
5501
5508 virtual AsciiChar const* getPluginNamespace() const noexcept = 0;
5509
5511 virtual ~IPluginCreatorV3One() = default;
5512
5513protected:
5516 IPluginCreatorV3One& operator=(IPluginCreatorV3One const&) & = default;
5517 IPluginCreatorV3One& operator=(IPluginCreatorV3One&&) & = default;
5518};
5519
5520} // namespace v_1_0
5521
5536
5546
5547} // namespace nvinfer1
5548
5552extern "C" TENSORRTAPI int32_t getInferLibMajorVersion() noexcept;
5556extern "C" TENSORRTAPI int32_t getInferLibMinorVersion() noexcept;
5560extern "C" TENSORRTAPI int32_t getInferLibPatchVersion() noexcept;
5564extern "C" TENSORRTAPI int32_t getInferLibBuildVersion() noexcept;
5565
5566#endif // NV_INFER_RUNTIME_H
TENSORRTAPI nvinfer1::IPluginRegistry * getPluginRegistry() noexcept
Return the plugin registry.
TENSORRTAPI nvinfer1::ILogger * getLogger() noexcept
Return the logger object.
TENSORRTAPI int32_t getInferLibMinorVersion() noexcept
Return the library minor version number.
TENSORRTAPI int32_t getInferLibMajorVersion() noexcept
Return the library major version number.
TENSORRTAPI int32_t getInferLibPatchVersion() noexcept
Return the library patch version number.
TENSORRTAPI int32_t getInferLibBuildVersion() noexcept
Return the library build version number.
#define TENSORRTAPI
Definition: NvInferRuntimeBase.h:70
#define NV_TENSORRT_VERSION
Definition: NvInferRuntimeBase.h:102
#define TRT_NODISCARD
A stand-in for [[nodiscard]] and [[nodiscard(REASON)]] that works with older compilers.
Definition: NvInferRuntimeBase.h:57
#define TRT_DEPRECATED
Definition: NvInferRuntimeBase.h:42
Structure to define the dimensions of a tensor.
Definition: NvInferRuntimeBase.h:224
static constexpr int32_t MAX_DIMS
The maximum rank (number of dimensions) supported for a tensor.
Definition: NvInferRuntimeBase.h:227
Analog of class Dims with expressions instead of constants for the dimensions.
Definition: NvInferRuntime.h:361
int32_t nbDims
The number of dimensions.
Definition: NvInferRuntime.h:363
An engine for executing inference on a built network, with functionally unsafe features.
Definition: NvInferRuntime.h:3144
int32_t getTensorBytesPerComponent(char const *tensorName) const noexcept
Return the number of bytes per component of an element, or -1 if the tensor is not vectorized or prov...
Definition: NvInferRuntime.h:3374
ISerializationConfig * createSerializationConfig() noexcept
Create a serialization configuration object.
Definition: NvInferRuntime.h:3734
char const * getIOTensorName(int32_t index) const noexcept
Return name of an IO tensor.
Definition: NvInferRuntime.h:3698
int64_t getWeightStreamingBudgetV2() const noexcept
Returns the current weight streaming device memory budget in bytes.
Definition: NvInferRuntime.h:3825
EngineCapability getEngineCapability() const noexcept
Determine what execution capability this engine has.
Definition: NvInferRuntime.h:3601
IErrorRecorder * getErrorRecorder() const noexcept
Get the ErrorRecorder assigned to this interface.
Definition: NvInferRuntime.h:3635
TensorFormat getTensorFormat(char const *tensorName, int32_t profileIndex) const noexcept
Return the tensor format of given profile, or TensorFormat::kLINEAR if the provided name does not map...
Definition: NvInferRuntime.h:3460
int64_t const * getProfileTensorValuesV2(char const *tensorName, int32_t profileIndex, OptProfileSelector select) const noexcept
Get the minimum / optimum / maximum values (not dimensions) for an input tensor given its name under ...
Definition: NvInferRuntime.h:3910
apiv::VCudaEngine * mImpl
Definition: NvInferRuntime.h:3945
IExecutionContext * createExecutionContext(ExecutionContextAllocationStrategy strategy=ExecutionContextAllocationStrategy::kSTATIC) noexcept
Create an execution context and specify the strategy for allocating internal activation memory.
Definition: NvInferRuntime.h:3218
char const * getTensorFormatDesc(char const *tensorName) const noexcept
Return the human readable description of the tensor format, or empty string if the provided name does...
Definition: NvInferRuntime.h:3484
Dims getProfileShape(char const *tensorName, int32_t profileIndex, OptProfileSelector select) const noexcept
Get the minimum / optimum / maximum dimensions for an input tensor given its name under an optimizati...
Definition: NvInferRuntime.h:3586
bool setWeightStreamingBudgetV2(int64_t gpuMemoryBudget) noexcept
Limit the maximum amount of GPU memory usable for network weights in bytes.
Definition: NvInferRuntime.h:3809
IExecutionContext * createExecutionContext(IRuntimeConfig *runtimeConfig) noexcept
Create an execution context with TensorRT JIT runtime config.
Definition: NvInferRuntime.h:3301
int32_t getNbAuxStreams() const noexcept
Return the number of auxiliary streams used by this engine.
Definition: NvInferRuntime.h:3724
int64_t getStreamableWeightsSize() const noexcept
Get the total size in bytes of all streamable weights.
Definition: NvInferRuntime.h:3770
DataType getTensorDataType(char const *tensorName) const noexcept
Determine the required data type for a buffer from its tensor name.
Definition: NvInferRuntime.h:3173
void setErrorRecorder(IErrorRecorder *recorder) noexcept
Set the ErrorRecorder for this interface.
Definition: NvInferRuntime.h:3620
TacticSources getTacticSources() const noexcept
return the tactic sources required by this engine.
Definition: NvInferRuntime.h:3651
TRT_NODISCARD char const * getAliasedInputTensor(char const *tensorName) const noexcept
Get the input tensor name that an output tensor should alias with.
Definition: NvInferRuntime.h:3289
IHostMemory * serializeWithConfig(ISerializationConfig &config) const noexcept
Serialize the network to a stream with the provided SerializationConfig.
Definition: NvInferRuntime.h:3754
int64_t getWeightStreamingAutomaticBudget() const noexcept
TensorRT automatically determines a device memory budget for the model to run. The budget is close to...
Definition: NvInferRuntime.h:3848
bool isDebugTensor(char const *name) const noexcept
Check if a tensor is marked as a debug tensor.
Definition: NvInferRuntime.h:3888
int32_t getTensorVectorizedDim(char const *tensorName, int32_t profileIndex) const noexcept
Return the dimension index that the buffer is vectorized of given profile, or -1 if the provided name...
Definition: NvInferRuntime.h:3540
char const * getName() const noexcept
Returns the name of the network associated with the engine.
Definition: NvInferRuntime.h:3555
ProfilingVerbosity getProfilingVerbosity() const noexcept
Return the ProfilingVerbosity the builder config was set to when the engine was built.
Definition: NvInferRuntime.h:3663
bool isShapeInferenceIO(char const *tensorName) const noexcept
True if tensor is required as input for shape calculations or is output from shape calculations.
Definition: NvInferRuntime.h:3256
int64_t getWeightStreamingScratchMemorySize() const noexcept
Returns the size of the scratch memory required by the current weight streaming budget.
Definition: NvInferRuntime.h:3874
int64_t getDeviceMemorySizeV2() const noexcept
Return the maximum device memory required by the context over all profiles.
Definition: NvInferRuntime.h:3329
int32_t getTensorVectorizedDim(char const *tensorName) const noexcept
Return the dimension index that the buffer is vectorized, or -1 if the provided name does not map to ...
Definition: NvInferRuntime.h:3524
int32_t getTensorComponentsPerElement(char const *tensorName, int32_t profileIndex) const noexcept
Return the number of components included in one element of given profile, or -1 if tensor is not vect...
Definition: NvInferRuntime.h:3431
int64_t getDeviceMemorySizeForProfileV2(int32_t profileIndex) const noexcept
Return the maximum device memory required by the context for a profile.
Definition: NvInferRuntime.h:3343
IRuntimeConfig * createRuntimeConfig() noexcept
Create a runtime config for TensorRT JIT. The caller is responsible for ownership of the returned IRu...
Definition: NvInferRuntime.h:3314
TensorFormat getTensorFormat(char const *tensorName) const noexcept
Return the tensor format, or TensorFormat::kLINEAR if the provided name does not map to an input or o...
Definition: NvInferRuntime.h:3446
IHostMemory * serialize() const noexcept
Serialize the network to a stream.
Definition: NvInferRuntime.h:3201
int64_t getEngineStat(EngineStat stat) const noexcept
Get engine statistics according to the given enum value.
Definition: NvInferRuntime.h:3938
TensorLocation getTensorLocation(char const *tensorName) const noexcept
Get whether an input or output tensor must be on GPU or CPU.
Definition: NvInferRuntime.h:3236
IEngineInspector * createEngineInspector() const noexcept
Create a new engine inspector which prints the layer information in an engine or an execution context...
Definition: NvInferRuntime.h:3673
int32_t getTensorBytesPerComponent(char const *tensorName, int32_t profileIndex) const noexcept
Return the number of bytes per component of an element given of given profile, or -1 if the tensor is...
Definition: NvInferRuntime.h:3392
HardwareCompatibilityLevel getHardwareCompatibilityLevel() const noexcept
Return the hardware compatibility level of this engine.
Definition: NvInferRuntime.h:3709
int32_t getNbOptimizationProfiles() const noexcept
Get the number of optimization profiles defined for this engine.
Definition: NvInferRuntime.h:3566
char const * getTensorFormatDesc(char const *tensorName, int32_t profileIndex) const noexcept
Return the human readable description of the tensor format of given profile, or empty string if the p...
Definition: NvInferRuntime.h:3507
TensorIOMode getTensorIOMode(char const *tensorName) const noexcept
Determine whether a tensor is an input or output tensor.
Definition: NvInferRuntime.h:3270
int32_t getNbLayers() const noexcept
Get the number of layers in the network.
Definition: NvInferRuntime.h:3187
int32_t getNbIOTensors() const noexcept
Return number of IO tensors.
Definition: NvInferRuntime.h:3686
virtual ~ICudaEngine() noexcept=0
int32_t getTensorComponentsPerElement(char const *tensorName) const noexcept
Return the number of components included in one element, or -1 if tensor is not vectorized or if the ...
Definition: NvInferRuntime.h:3413
bool isRefittable() const noexcept
Return true if an engine can be refit.
Definition: NvInferRuntime.h:3353
An IDimensionExpr represents an integer expression constructed from constants, input dimensions,...
Definition: NvInferRuntime.h:239
bool isConstant() const noexcept
Return true if expression is a build-time constant.
Definition: NvInferRuntime.h:244
bool isSizeTensor() const noexcept
Return true if this denotes the value of a size tensor.
Definition: NvInferRuntime.h:270
virtual ~IDimensionExpr() noexcept=0
apiv::VDimensionExpr * mImpl
Definition: NvInferRuntime.h:261
int64_t getConstantValue() const noexcept
Get the value of the constant.
Definition: NvInferRuntime.h:255
An engine inspector which prints out the layer information of an engine or an execution context.
Definition: NvInferRuntime.h:5069
char const * getLayerInformation(int32_t layerIndex, LayerInformationFormat format) const noexcept
Get a string describing the information about a specific layer in the current engine or the execution...
Definition: NvInferRuntime.h:5122
IErrorRecorder * getErrorRecorder() const noexcept
Get the ErrorRecorder assigned to this interface.
Definition: NvInferRuntime.h:5179
void setErrorRecorder(IErrorRecorder *recorder) noexcept
Set the ErrorRecorder for this interface.
Definition: NvInferRuntime.h:5164
virtual ~IEngineInspector() noexcept=0
IExecutionContext const * getExecutionContext() const noexcept
Get the context currently being inspected.
Definition: NvInferRuntime.h:5097
apiv::VEngineInspector * mImpl
Definition: NvInferRuntime.h:5185
char const * getEngineInformation(LayerInformationFormat format) const noexcept
Get a string describing the information about all the layers in the current engine or the execution c...
Definition: NvInferRuntime.h:5145
Context for executing inference using an engine, with functionally unsafe features.
Definition: NvInferRuntime.h:4090
IOutputAllocator * getOutputAllocator(char const *tensorName) const noexcept
Get output allocator associated with output tensor of given name, or nullptr if the provided name doe...
Definition: NvInferRuntime.h:4706
IErrorRecorder * getErrorRecorder() const noexcept
Get the ErrorRecorder assigned to this interface.
Definition: NvInferRuntime.h:4354
bool reportToProfiler() const noexcept
Calculate layer timing info for the current optimization profile in IExecutionContext and update the ...
Definition: NvInferRuntime.h:4473
void setDeviceMemory(void *memory) noexcept
Set the device memory for use by this execution context.
Definition: NvInferRuntime.h:4192
bool setTensorDebugState(char const *name, bool flag) noexcept
Set debug state of tensor given the tensor name.
Definition: NvInferRuntime.h:4943
char const * getName() const noexcept
Return the name of the execution context.
Definition: NvInferRuntime.h:4166
IGpuAllocator * getTemporaryStorageAllocator() const noexcept
Get allocator set by setTemporaryStorageAllocator.
Definition: NvInferRuntime.h:4754
void setEnqueueEmitsProfile(bool enqueueEmitsProfile) noexcept
Set whether enqueue emits layer timing to the profiler.
Definition: NvInferRuntime.h:4431
bool setUnfusedTensorsDebugState(bool flag) noexcept
Turn the debug state of unfused tensors on or off.
Definition: NvInferRuntime.h:4994
Dims getTensorShape(char const *tensorName) const noexcept
Return the shape of the given input or output.
Definition: NvInferRuntime.h:4304
bool getDebugState(char const *name) const noexcept
Get the debug state.
Definition: NvInferRuntime.h:4955
bool setInputShape(char const *tensorName, Dims const &dims) noexcept
Set shape of given input.
Definition: NvInferRuntime.h:4267
bool executeV2(void *const *bindings) noexcept
Synchronously execute a network.
Definition: NvInferRuntime.h:4371
bool getEnqueueEmitsProfile() const noexcept
Get the enqueueEmitsProfile state.
Definition: NvInferRuntime.h:4443
void const * getTensorAddress(char const *tensorName) const noexcept
Get memory address bound to given input or output tensor, or nullptr if the provided name does not ma...
Definition: NvInferRuntime.h:4534
bool setOutputAllocator(char const *tensorName, IOutputAllocator *outputAllocator) noexcept
Set output allocator to use for output tensor of given name. Pass nullptr to outputAllocator to unset...
Definition: NvInferRuntime.h:4693
bool setOptimizationProfileAsync(int32_t profileIndex, cudaStream_t stream) noexcept
Select an optimization profile for the current context with async semantics.
Definition: NvInferRuntime.h:4415
apiv::VExecutionContext * mImpl
Definition: NvInferRuntime.h:5028
bool setOutputTensorAddress(char const *tensorName, void *data) noexcept
Set the memory address for a given output tensor.
Definition: NvInferRuntime.h:4557
void setPersistentCacheLimit(size_t size) noexcept
Set the maximum size for persistent cache usage.
Definition: NvInferRuntime.h:4807
virtual ~IExecutionContext() noexcept=0
size_t getPersistentCacheLimit() const noexcept
Get the maximum size for persistent cache usage.
Definition: NvInferRuntime.h:4818
bool setAllTensorsDebugState(bool flag) noexcept
Turn the debug state of all debug tensors on or off.
Definition: NvInferRuntime.h:4978
ICudaEngine const & getEngine() const noexcept
Get the associated engine.
Definition: NvInferRuntime.h:4142
ProfilingVerbosity getNvtxVerbosity() const noexcept
Get the NVTX verbosity of the execution context.
Definition: NvInferRuntime.h:4854
size_t updateDeviceMemorySizeForShapes() noexcept
Recompute the internal activation buffer sizes based on the current input shapes, and return the tota...
Definition: NvInferRuntime.h:4648
void setAuxStreams(cudaStream_t *auxStreams, int32_t nbStreams) noexcept
Set the auxiliary streams that TensorRT should launch kernels on in the next enqueueV3() call.
Definition: NvInferRuntime.h:4902
int64_t getMaxOutputSize(char const *tensorName) const noexcept
Get upper bound on an output tensor's size, in bytes, based on the current optimization profile and i...
Definition: NvInferRuntime.h:4724
int32_t inferShapes(int32_t nbMaxNames, char const **tensorNames) noexcept
Run shape calculations.
Definition: NvInferRuntime.h:4631
bool setDebugListener(IDebugListener *listener) noexcept
Set DebugListener for this execution context.
Definition: NvInferRuntime.h:4914
bool setTensorAddress(char const *tensorName, void *data) noexcept
Set memory address for given input or output tensor.
Definition: NvInferRuntime.h:4517
bool setTemporaryStorageAllocator(IGpuAllocator *allocator) noexcept
Specify allocator to use for internal temporary storage.
Definition: NvInferRuntime.h:4744
void * getOutputTensorAddress(char const *tensorName) const noexcept
Get memory address for given output.
Definition: NvInferRuntime.h:4598
bool enqueueV3(cudaStream_t stream) noexcept
Enqueue inference on a stream.
Definition: NvInferRuntime.h:4791
IDebugListener * getDebugListener() noexcept
Get the DebugListener of this execution context.
Definition: NvInferRuntime.h:4924
int32_t getOptimizationProfile() const noexcept
Get the index of the currently selected optimization profile.
Definition: NvInferRuntime.h:4249
bool setInputTensorAddress(char const *tensorName, void const *data) noexcept
Set memory address for given input.
Definition: NvInferRuntime.h:4579
bool getDebugSync() const noexcept
Get the debug sync flag.
Definition: NvInferRuntime.h:4112
bool setInputConsumedEvent(cudaEvent_t event) noexcept
Mark input as consumed.
Definition: NvInferRuntime.h:4664
Dims getTensorStrides(char const *tensorName) const noexcept
Return the strides of the buffer for the given tensor name.
Definition: NvInferRuntime.h:4234
bool setNvtxVerbosity(ProfilingVerbosity verbosity) noexcept
Set the verbosity of the NVTX markers in the execution context.
Definition: NvInferRuntime.h:4842
IProfiler * getProfiler() const noexcept
Get the profiler.
Definition: NvInferRuntime.h:4132
void setErrorRecorder(IErrorRecorder *recorder) noexcept
Set the ErrorRecorder for this interface.
Definition: NvInferRuntime.h:4339
bool setCommunicator(void *communicator) noexcept
Set the NCCL communicator for the execution context.
Definition: NvInferRuntime.h:5022
void setDeviceMemoryV2(void *memory, int64_t size) noexcept
Set the device memory and its corresponding size for use by this execution context.
Definition: NvInferRuntime.h:4213
bool allInputDimensionsSpecified() const noexcept
Whether all dynamic dimensions of input tensors have been specified.
Definition: NvInferRuntime.h:4320
bool getUnfusedTensorsDebugState() const noexcept
Get the debug state of unfused tensors.
Definition: NvInferRuntime.h:5004
void setProfiler(IProfiler *profiler) noexcept
Set the profiler.
Definition: NvInferRuntime.h:4122
void setName(char const *name) noexcept
Set the name of the execution context.
Definition: NvInferRuntime.h:4156
cudaEvent_t getInputConsumedEvent() const noexcept
The event associated with consuming the input.
Definition: NvInferRuntime.h:4674
IRuntimeConfig * getRuntimeConfig() const noexcept
Get the runtime config object used during execution context creation.
Definition: NvInferRuntime.h:4965
Object for constructing IDimensionExpr.
Definition: NvInferRuntime.h:296
IDimensionExpr const * operation(DimensionOperation op, IDimensionExpr const &first, IDimensionExpr const &second) noexcept
Get the operation.
Definition: NvInferRuntime.h:312
IDimensionExpr const * constant(int64_t value) noexcept
Return pointer to IDimensionExpr for given value.
Definition: NvInferRuntime.h:301
apiv::VExprBuilder * mImpl
Definition: NvInferRuntime.h:319
virtual ~IExprBuilder() noexcept=0
Class to handle library allocated memory that is accessible to the user.
Definition: NvInferRuntime.h:150
void * data() const noexcept
A pointer to the raw data that is owned by the library.
Definition: NvInferRuntime.h:155
virtual ~IHostMemory() noexcept=0
DataType type() const noexcept
The type of the memory that was allocated.
Definition: NvInferRuntime.h:167
std::size_t size() const noexcept
The size in bytes of the data that was allocated.
Definition: NvInferRuntime.h:161
apiv::VHostMemory * mImpl
Definition: NvInferRuntime.h:173
Forward declaration of IEngineInspector for use by other interfaces.
Definition: NvInferRuntime.h:51
INoCopy & operator=(INoCopy &&other)=delete
INoCopy(INoCopy const &other)=delete
INoCopy(INoCopy &&other)=delete
virtual ~INoCopy()=default
INoCopy & operator=(INoCopy const &other)=delete
Optimization profile for dynamic input dimensions and shape tensors.
Definition: NvInferRuntime.h:2667
apiv::VOptimizationProfile * mImpl
Definition: NvInferRuntime.h:2872
Dims getDimensions(char const *inputName, OptProfileSelector select) const noexcept
Get the minimum / optimum / maximum dimensions for a dynamic input tensor.
Definition: NvInferRuntime.h:2708
float getExtraMemoryTarget() const noexcept
Get the extra memory target that has been defined for this profile.
Definition: NvInferRuntime.h:2751
cudaStream_t getProfileStream() const noexcept
Get the CUDA stream set for this optimization profile.
Definition: NvInferRuntime.h:2866
bool setExtraMemoryTarget(float target) noexcept
Set a target for extra GPU memory that may be used by this profile.
Definition: NvInferRuntime.h:2739
bool setDimensions(char const *inputName, OptProfileSelector select, Dims const &dims) noexcept
Set the minimum / optimum / maximum dimensions for a dynamic input tensor.
Definition: NvInferRuntime.h:2696
virtual ~IOptimizationProfile() noexcept=0
void setProfileStream(cudaStream_t stream) noexcept
Set the CUDA stream that is used to profile this optimization profile.
Definition: NvInferRuntime.h:2854
bool isValid() const noexcept
Check whether the optimization profile can be passed to an IBuilderConfig object.
Definition: NvInferRuntime.h:2768
int64_t const * getShapeValuesV2(char const *inputName, OptProfileSelector select) const noexcept
Get the minimum / optimum / maximum values for an input shape tensor.
Definition: NvInferRuntime.h:2828
bool setShapeValuesV2(char const *inputName, OptProfileSelector select, int64_t const *values, int32_t nbValues) noexcept
Set the minimum / optimum / maximum values for an input shape tensor.
Definition: NvInferRuntime.h:2815
int32_t getNbShapeValues(char const *inputName) const noexcept
Get the number of values for an input shape tensor.
Definition: NvInferRuntime.h:2721
Single registration point for all plugins in an application. It is used to find plugin implementation...
Definition: NvInferRuntimeCommon.h:56
virtual bool registerCreator(IPluginCreatorInterface &creator, AsciiChar const *const pluginNamespace) noexcept=0
Register a plugin creator. Returns false if a plugin creator with the same type is already registered...
Interface for plugins to access per context resources provided by TensorRT.
Definition: NvInferRuntime.h:803
virtual IErrorRecorder * getErrorRecorder() const noexcept=0
Get the error recorder associated with the resource context.
IPluginResourceContext & operator=(IPluginResourceContext const &) &=default
virtual IGpuAllocator * getGpuAllocator() const noexcept=0
Get the GPU allocator associated with the resource context.
Similar to IPluginV2Ext, but with support for dynamic shapes.
Definition: NvInferRuntime.h:418
IPluginV2DynamicExt * clone() const noexcept override=0
Clone the plugin object. This copies over internal plugin parameters as well and returns a new plugin...
virtual ~IPluginV2DynamicExt() noexcept
Definition: NvInferRuntime.h:567
Plugin class for user-implemented layers.
Definition: NvInferRuntimePlugin.h:474
Updates weights in an engine.
Definition: NvInferRuntime.h:2238
bool refitCudaEngineAsync(cudaStream_t stream) noexcept
Enqueue weights refitting of the associated engine on the given stream.
Definition: NvInferRuntime.h:2577
int32_t getMaxThreads() const noexcept
get the maximum number of threads that can be used by the refitter.
Definition: NvInferRuntime.h:2447
TensorLocation getWeightsLocation(char const *weightsName) const noexcept
Get location for the weights associated with the given name.
Definition: NvInferRuntime.h:2506
bool setNamedWeights(char const *name, Weights weights) noexcept
Specify new weights of given name.
Definition: NvInferRuntime.h:2371
bool releaseRefitResources() noexcept
Release resources cached for the engine associated with this refitter.
Definition: NvInferRuntime.h:2609
int32_t getAllWeights(int32_t size, char const **weightsNames) noexcept
Get names of all weights that could be refit.
Definition: NvInferRuntime.h:2407
virtual ~IRefitter() noexcept=0
ILogger * getLogger() const noexcept
get the logger with which the refitter was created
Definition: NvInferRuntime.h:2417
bool refitCudaEngine() noexcept
Refits associated engine.
Definition: NvInferRuntime.h:2274
int32_t getMissingWeights(int32_t size, char const **weightsNames) noexcept
Get names of missing weights.
Definition: NvInferRuntime.h:2391
int32_t getMissing(int32_t size, char const **layerNames, WeightsRole *roles) noexcept
Get description of missing weights.
Definition: NvInferRuntime.h:2295
Weights getNamedWeights(char const *weightsName) const noexcept
Get weights associated with the given name.
Definition: NvInferRuntime.h:2490
bool unsetNamedWeights(char const *weightsName) noexcept
Unset weights associated with the given name.
Definition: NvInferRuntime.h:2522
Weights getWeightsPrototype(char const *weightsName) const noexcept
Get the Weights prototype associated with the given name.
Definition: NvInferRuntime.h:2595
bool setMaxThreads(int32_t maxThreads) noexcept
Set the maximum number of threads.
Definition: NvInferRuntime.h:2433
bool setNamedWeights(char const *name, Weights weights, TensorLocation location) noexcept
Specify new weights on a specified device of given name.
Definition: NvInferRuntime.h:2474
void setWeightsValidation(bool weightsValidation) noexcept
Set whether to validate weights during refitting.
Definition: NvInferRuntime.h:2538
apiv::VRefitter * mImpl
Definition: NvInferRuntime.h:2615
int32_t getAll(int32_t size, char const **layerNames, WeightsRole *roles) noexcept
Get description of all weights that could be refit.
Definition: NvInferRuntime.h:2312
bool getWeightsValidation() const noexcept
Get whether to validate weights values during refitting.
Definition: NvInferRuntime.h:2546
void setErrorRecorder(IErrorRecorder *recorder) noexcept
Set the ErrorRecorder for this interface.
Definition: NvInferRuntime.h:2331
IErrorRecorder * getErrorRecorder() const noexcept
Get the ErrorRecorder assigned to this interface.
Definition: NvInferRuntime.h:2346
A class for runtime configuration. This class is used during execution context creation.
Definition: NvInferRuntime.h:3080
apiv::VRuntimeConfig * mImpl
Definition: NvInferRuntime.h:3106
virtual ~IRuntimeConfig() noexcept=0
ExecutionContextAllocationStrategy getExecutionContextAllocationStrategy() const noexcept
Get the execution context allocation strategy.
Definition: NvInferRuntime.h:3099
Allows a serialized functionally unsafe engine to be deserialized.
Definition: NvInferRuntime.h:1901
bool setMaxThreads(int32_t maxThreads) noexcept
Set the maximum number of threads.
Definition: NvInferRuntime.h:2080
IRuntime * loadRuntime(char const *path) noexcept
Load IRuntime from the file.
Definition: NvInferRuntime.h:2196
bool getEngineHostCodeAllowed() const noexcept
Get whether the runtime is allowed to deserialize engines with host executable code.
Definition: NvInferRuntime.h:2218
TempfileControlFlags getTempfileControlFlags() const noexcept
Get the tempfile control flags for this runtime.
Definition: NvInferRuntime.h:2168
void setEngineHostCodeAllowed(bool allowed) noexcept
Set whether the runtime is allowed to deserialize engines with host executable code.
Definition: NvInferRuntime.h:2208
void setTemporaryDirectory(char const *path) noexcept
Set the directory that will be used by this runtime for temporary files.
Definition: NvInferRuntime.h:2129
bool setDLAWorkspaceAllocationStrategy(DLAWorkspaceAllocationStrategy strategy) noexcept
Sets the strategy used for DLA workspace allocation by subsequent engine deserializations.
Definition: NvInferRuntime.h:1954
IPluginRegistry & getPluginRegistry() noexcept
Get the local plugin registry that can be used by the runtime.
Definition: NvInferRuntime.h:2178
TRT_NODISCARD DLAWorkspaceAllocationStrategy getDLAWorkspaceAllocationStrategy() const noexcept
Returns the DLA workspace allocation strategy used for subsequent engine deserializations.
Definition: NvInferRuntime.h:1964
int32_t getNbDLACores() const noexcept
Returns number of DLA hardware cores accessible or 0 if DLA is unavailable.
Definition: NvInferRuntime.h:1937
ICudaEngine * deserializeCudaEngine(void const *blob, std::size_t size) noexcept
Deserialize an engine from host memory.
Definition: NvInferRuntime.h:2032
virtual ~IRuntime() noexcept=0
void setTempfileControlFlags(TempfileControlFlags flags) noexcept
Set the tempfile control flags for this runtime.
Definition: NvInferRuntime.h:2156
int32_t getDLACore() const noexcept
Get the DLA core that the engine executes on.
Definition: NvInferRuntime.h:1929
void setGpuAllocator(IGpuAllocator *allocator) noexcept
Set the GPU allocator.
Definition: NvInferRuntime.h:1980
IErrorRecorder * getErrorRecorder() const noexcept
get the ErrorRecorder assigned to this interface.
Definition: NvInferRuntime.h:2014
ICudaEngine * deserializeCudaEngine(IStreamReaderV2 &streamReader)
Deserialize an engine from a stream. IStreamReaderV2 is expected to support reading to both host and ...
Definition: NvInferRuntime.h:2055
ILogger * getLogger() const noexcept
get the logger with which the runtime was created
Definition: NvInferRuntime.h:2065
int32_t getMaxThreads() const noexcept
Get the maximum number of threads that can be used by the runtime.
Definition: NvInferRuntime.h:2094
char const * getTemporaryDirectory() const noexcept
Get the directory that will be used by this runtime for temporary files.
Definition: NvInferRuntime.h:2140
void setErrorRecorder(IErrorRecorder *recorder) noexcept
Set the ErrorRecorder for this interface.
Definition: NvInferRuntime.h:1999
Holds properties for configuring an engine to serialize the binary.
Definition: NvInferRuntime.h:2972
bool clearFlag(SerializationFlag serializationFlag) noexcept
clear a serialization flag.
Definition: NvInferRuntime.h:3011
virtual ~ISerializationConfig() noexcept=0
bool setFlag(SerializationFlag serializationFlag) noexcept
Set a serialization flag.
Definition: NvInferRuntime.h:3023
SerializationFlags getFlags() const noexcept
Get the serialization flags for this config.
Definition: NvInferRuntime.h:2999
bool getFlag(SerializationFlag serializationFlag) const noexcept
Returns true if the serialization flag is set.
Definition: NvInferRuntime.h:3035
apiv::VSerializationConfig * mImpl
Definition: NvInferRuntime.h:3041
An Interface class for version control.
Definition: NvInferRuntimeBase.h:284
Version information associated with a TRT interface.
Definition: NvInferRuntimeBase.h:249
Register the plugin creator to the registry The static registry object will be instantiated when the ...
Definition: NvInferRuntime.h:5257
PluginRegistrar()
Definition: NvInferRuntime.h:5259
An array of weights used as a layer parameter.
Definition: NvInferRuntime.h:132
DataType type
The type of the weights.
Definition: NvInferRuntime.h:134
int64_t count
The number of weights in the array.
Definition: NvInferRuntime.h:136
void const * values
The weight values, in a contiguous array.
Definition: NvInferRuntime.h:135
Definition: NvInferRuntime.h:4041
virtual bool processDebugTensor(void const *addr, TensorLocation location, DataType type, Dims const &shape, char const *name, cudaStream_t stream)=0
Callback function that is called when a debug tensor’s value is updated and the debug state of the te...
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:4046
~IDebugListener() override=default
Definition: NvInferRuntimeBase.h:421
Definition: NvInferRuntime.h:1691
virtual void * allocateAsync(uint64_t const size, uint64_t const alignment, AllocatorFlags const flags, cudaStream_t) noexcept
A thread-safe callback implemented by the application to handle stream-ordered acquisition of GPU mem...
Definition: NvInferRuntime.h:1813
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:1854
virtual TRT_DEPRECATED bool deallocate(void *const memory) noexcept=0
A thread-safe callback implemented by the application to handle release of GPU memory.
~IGpuAllocator() override=default
virtual void * reallocate(void *const, uint64_t, uint64_t) noexcept
A thread-safe callback implemented by the application to resize an existing allocation.
Definition: NvInferRuntime.h:1760
virtual TRT_DEPRECATED void * allocate(uint64_t const size, uint64_t const alignment, AllocatorFlags const flags) noexcept=0
A thread-safe callback implemented by the application to handle acquisition of GPU memory.
virtual bool deallocateAsync(void *const memory, cudaStream_t) noexcept
A thread-safe callback implemented by the application to handle stream-ordered release of GPU memory.
Definition: NvInferRuntime.h:1846
Definition: NvInferRuntime.h:5323
bool deallocateAsync(void *const memory, cudaStream_t) noexcept override=0
A thread-safe callback implemented by the application to handle stream-ordered asynchronous release o...
void * allocateAsync(uint64_t const size, uint64_t const alignment, AllocatorFlags const flags, cudaStream_t) noexcept override=0
A thread-safe callback implemented by the application to handle stream-ordered asynchronous acquisiti...
TRT_DEPRECATED void * allocate(uint64_t const size, uint64_t const alignment, AllocatorFlags const flags) noexcept override
A thread-safe callback implemented by the application to handle acquisition of GPU memory.
Definition: NvInferRuntime.h:5410
TRT_DEPRECATED bool deallocate(void *const memory) noexcept override
A thread-safe callback implemented by the application to handle release of GPU memory.
Definition: NvInferRuntime.h:5434
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:5442
~IGpuAsyncAllocator() override=default
A virtual base class to find a logger. Allows a plugin to find an instance of a logger if it needs to...
Definition: NvInferRuntime.h:5289
virtual ILogger * findLogger()=0
Get the logger used by the engine or execution context which called the plugin method.
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:5294
~ILoggerFinder() override=default
Protected: TRT owns ILoggerFinder instances and passes non-owning pointers to plugins.
Application-implemented logging interface for the builder, refitter and runtime.
Definition: NvInferRuntime.h:1614
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:1619
~ILogger() override=default
Severity
The severity corresponding to a log message.
Definition: NvInferRuntime.h:1630
virtual void log(Severity severity, AsciiChar const *msg) noexcept=0
A callback implemented by the application to handle logging messages;.
Definition: NvInferRuntime.h:3953
virtual TRT_DEPRECATED void * reallocateOutput(char const *, void *, uint64_t, uint64_t) noexcept
Return a pointer to memory for an output tensor, or nullptr if memory cannot be allocated....
Definition: NvInferRuntime.h:3982
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:3958
virtual void * reallocateOutputAsync(char const *tensorName, void *currentMemory, uint64_t size, uint64_t alignment, cudaStream_t)
Return a pointer to memory for an output tensor, or nullptr if memory cannot be allocated....
Definition: NvInferRuntime.h:4010
virtual void notifyShape(char const *tensorName, Dims const &dims) noexcept=0
Called by TensorRT when the shape of the output tensor is known.
Definition: NvInferPluginBase.h:141
Definition: NvInferPluginBase.h:193
Definition: NvInferRuntime.h:5449
virtual PluginFieldCollection const * getFieldNames() noexcept=0
Return a list of fields that need to be passed to createPlugin() when creating a plugin for use in th...
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:5454
virtual IPluginV3 * createPlugin(AsciiChar const *name, PluginFieldCollection const *fc, TensorRTPhase phase) noexcept=0
Return a plugin object. Return nullptr in case of error.
Definition: NvInferPluginBase.h:206
Definition: NvInferRuntime.h:874
virtual int32_t getFormatCombinationLimit() noexcept
Return the maximum number of format combinations that will be timed by TensorRT during the build phas...
Definition: NvInferRuntime.h:1078
virtual int32_t getNbOutputs() const noexcept=0
Get the number of outputs from the plugin.
virtual int32_t configurePlugin(DynamicPluginTensorDesc const *in, int32_t nbInputs, DynamicPluginTensorDesc const *out, int32_t nbOutputs) noexcept=0
Configure the plugin.
virtual int32_t getOutputDataTypes(DataType *outputTypes, int32_t nbOutputs, DataType const *inputTypes, int32_t nbInputs) const noexcept=0
Provide the data types of the plugin outputs if the input tensors have the data types provided.
virtual int32_t getNbTactics() noexcept
Query for the number of custom tactics the plugin intends to use.
Definition: NvInferRuntime.h:1054
virtual char const * getMetadataString() noexcept
Query for a string representing the configuration of the plugin. May be called anytime after plugin c...
Definition: NvInferRuntime.h:1089
virtual char const * getTimingCacheID() noexcept
Called to query the suffix to use for the timing cache ID. May be called anytime after plugin creatio...
Definition: NvInferRuntime.h:1070
virtual bool supportsFormatCombination(int32_t pos, DynamicPluginTensorDesc const *inOut, int32_t nbInputs, int32_t nbOutputs) noexcept=0
Return true if plugin supports the format and datatype for the input/output indexed by pos.
virtual int32_t getValidTactics(int32_t *, int32_t) noexcept
Query for any custom tactics that the plugin intends to use.
Definition: NvInferRuntime.h:1046
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:886
virtual int32_t getOutputShapes(DimsExprs const *inputs, int32_t nbInputs, DimsExprs const *shapeInputs, int32_t nbShapeInputs, DimsExprs *outputs, int32_t nbOutputs, IExprBuilder &exprBuilder) noexcept=0
Provide expressions for computing dimensions of the output tensors from dimensions of the input tenso...
Definition: NvInferRuntime.h:831
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:836
virtual AsciiChar const * getPluginName() const noexcept=0
Return the plugin name. Should match the plugin name returned by the corresponding plugin creator.
Definition: NvInferRuntime.h:1096
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:1101
virtual int32_t onShapeChange(PluginTensorDesc const *in, int32_t nbInputs, PluginTensorDesc const *out, int32_t nbOutputs) noexcept=0
Called when a plugin is being prepared for execution for specific dimensions. This could happen multi...
virtual PluginFieldCollection const * getFieldsToSerialize() noexcept=0
Get the plugin fields which should be serialized.
virtual int32_t setTactic(int32_t) noexcept
Set the tactic to be used in the subsequent call to enqueue(). If no custom tactics were advertised,...
Definition: NvInferRuntime.h:1113
virtual int32_t enqueue(PluginTensorDesc const *inputDesc, PluginTensorDesc const *outputDesc, void const *const *inputs, void *const *outputs, void *workspace, cudaStream_t stream) noexcept=0
Execute the layer.
virtual IPluginV3 * attachToContext(IPluginResourceContext *context) noexcept=0
Clone the plugin, attach the cloned plugin object to a execution context and grant the cloned plugin ...
Definition: NvInferRuntime.h:1286
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:1291
~IProfiler() override=default
virtual void reportLayerTime(char const *layerName, float ms) noexcept=0
Layer time reporting callback.
Definition: NvInferRuntime.h:619
~IStreamReader() override=default
IStreamReader & operator=(IStreamReader const &) &=default
IStreamReader & operator=(IStreamReader &&) &=default
virtual int64_t read(void *destination, int64_t nbBytes)=0
Read the next number of bytes in the stream.
IStreamReader(IStreamReader &&)=default
IStreamReader(IStreamReader const &)=default
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:631
Definition: NvInferRuntime.h:731
IStreamReaderV2 & operator=(IStreamReaderV2 const &) &=default
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:743
IStreamReaderV2(IStreamReaderV2 &&)=default
~IStreamReaderV2() override=default
virtual int64_t read(void *destination, int64_t nbBytes, cudaStream_t stream) noexcept=0
Read the next number of bytes in the stream asynchronously.
IStreamReaderV2(IStreamReaderV2 const &)=default
virtual bool seek(int64_t offset, SeekPosition where) noexcept=0
Sets the position of the stream to the given offset.
IStreamReaderV2 & operator=(IStreamReaderV2 &&) &=default
Definition: NvInferRuntime.h:654
IStreamWriter & operator=(IStreamWriter const &) &=default
IStreamWriter(IStreamWriter &&)=default
virtual int64_t write(void const *data, int64_t nbBytes)=0
write nbBytes of data into the stream.
IStreamWriter(IStreamWriter const &)=default
IStreamWriter & operator=(IStreamWriter &&) &=default
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:666
~IStreamWriter() override=default
Definition: NvInferRuntime.h:1193
InterfaceInfo getInterfaceInfo() const noexcept override
Return version information associated with this interface. Applications must not override this method...
Definition: NvInferRuntime.h:1195
virtual int32_t getAliasedInput(int32_t) noexcept
Communicates to TensorRT that the output at the specified output index is aliased to the input at the...
Definition: NvInferRuntime.h:1229
IRefitter * createInferRefitter(ICudaEngine &engine, ILogger &logger) noexcept
Create an instance of an IRefitter class.
Definition: NvInferRuntime.h:5237
IRuntime * createInferRuntime(ILogger &logger) noexcept
Create an instance of an IRuntime class.
Definition: NvInferRuntime.h:5226
The TensorRT API version 1 namespace.
Definition: NvInferSafePlugin.h:33
uint32_t TacticSources
Represents a collection of one or more TacticSource values combine using bitwise-OR operations.
Definition: NvInferRuntime.h:2910
v_1_0::IOutputAllocator IOutputAllocator
Definition: NvInferRuntime.h:4036
EngineCapability
List of supported engine capability flows.
Definition: NvInferRuntime.h:76
DimensionOperation
An operation on two IDimensionExpr, which represent integer expressions used in dimension computation...
Definition: NvInferRuntime.h:189
@ kCEIL_DIV
Division rounding up.
v_1_0::IPluginV3OneCore IPluginV3OneCore
Definition: NvInferRuntime.h:1246
TensorIOMode
Definition of tensor IO Mode.
Definition: NvInferRuntimeBase.h:664
HardwareCompatibilityLevel
Describes requirements of compatibility with GPU architectures other than that of the GPU on which th...
Definition: NvInfer.h:10447
SerializationFlag
List of valid flags that the engine can enable when serializing the bytes.
Definition: NvInferRuntime.h:2951
@ kEXCLUDE_WEIGHTS
Exclude the weights that can be refitted.
@ kINCLUDE_REFIT
Remain refittable if originally so.
DLAWorkspaceAllocationStrategy
Describes how DLA workspace memory is allocated.
Definition: NvInferRuntime.h:1576
v_1_0::IStreamWriter IStreamWriter
Definition: NvInferRuntime.h:710
v_1_0::IProfiler IProfiler
Definition: NvInferRuntime.h:1320
SeekPosition
Controls the seek mode of IStreamReaderV2.
Definition: NvInferRuntime.h:717
@ kSET
From the beginning of the file.
@ kCUR
From the current position of the file.
@ kEND
From the tail of the file.
v_1_0::IStreamReaderV2 IStreamReaderV2
Definition: NvInferRuntime.h:787
uint32_t TempfileControlFlags
Represents a collection of one or more TempfileControlFlag values combined using bitwise-OR operation...
Definition: NvInferRuntime.h:1398
EngineStat
The kind of engine statistics that queried from the ICudaEngine.
Definition: NvInferRuntime.h:3121
@ kTOTAL_WEIGHTS_SIZE
Return the total weight size in bytes.
@ kSTRIPPED_WEIGHTS_SIZE
Return the stripped weight size in bytes for engines built with BuilderFlag::kSTRIP_PLAN.
v_1_0::IGpuAllocator IGpuAllocator
Definition: NvInferRuntime.h:1890
v_1_0::ILogger ILogger
Definition: NvInferRuntimeBase.h:125
char_t AsciiChar
Definition: NvInferRuntimeBase.h:116
TensorRTPhase
Indicates a phase of operation of TensorRT.
Definition: NvInferPluginBase.h:116
@ kV2_DYNAMICEXT
IPluginV2DynamicExt.
DataType
The type of weights and tensors. The datatypes other than kBOOL, kINT32, and kINT64 are "activation d...
Definition: NvInferRuntimeBase.h:151
DeviceType
The device that this layer/network will execute on.
Definition: NvInferRuntime.h:1352
@ kSCALE
Scale layer.
@ kCONSTANT
Constant layer.
@ kDEFAULT
Similar to ONNX Gather.
v_1_0::IDebugListener IDebugListener
Definition: NvInferRuntime.h:4077
TempfileControlFlag
Flags used to control TensorRT's behavior when creating executable temporary files.
Definition: NvInferRuntime.h:1375
@ kALLOW_IN_MEMORY_FILES
Allow creating and loading files in-memory (or unnamed files).
WeightsRole
How a layer uses particular Weights.
Definition: NvInferRuntime.h:1330
@ kSHIFT
shift part of IScaleLayer
@ kANY
Any other weights role.
@ kBIAS
bias for IConvolutionLayer or IDeconvolutionLayer
@ kKERNEL
kernel for IConvolutionLayer or IDeconvolutionLayer
ProfilingVerbosity
List of verbosity levels of layer information exposed in NVTX annotations and in IEngineInspector.
Definition: NvInferRuntime.h:2922
@ kLAYER_NAMES_ONLY
Print only the layer names. This is the default setting.
@ kDETAILED
Print detailed layer information including layer names and layer parameters.
TacticSource
List of tactic sources for TensorRT.
Definition: NvInferRuntime.h:2886
TensorFormat PluginFormat
PluginFormat is reserved for backward compatibility.
Definition: NvInferRuntimePlugin.h:54
v_1_0::IPluginV3OneRuntime IPluginV3OneRuntime
Definition: NvInferRuntime.h:1270
@ kSUB
Subtract the second element from the first.
@ kSUM
Sum of the two elements.
@ kPROD
Product of the two elements.
@ kFLOOR_DIV
Floor division of the first element by the second.
@ kEQUAL
Check if two elements are equal.
@ kMIN
Minimum of the two elements.
@ kLESS
Check if element in first tensor is less than corresponding element in second tensor.
uint32_t SerializationFlags
Represents one or more SerializationFlag values using binary OR operations, e.g., 1U << Serialization...
Definition: NvInferRuntime.h:2941
@ kLINEAR
Supports linear (1D), bilinear (2D), and trilinear (3D) interpolation.
v_1_0::IPluginV3OneBuild IPluginV3OneBuild
Definition: NvInferRuntime.h:1258
TensorFormat
Format of the input/output tensors.
Definition: NvInferRuntime.h:1432
ExecutionContextAllocationStrategy
Different memory allocation behaviors for IExecutionContext.
Definition: NvInferRuntime.h:3058
@ kSTATIC
Default static allocation with the maximum size across all profiles.
@ kUSER_MANAGED
The user supplies custom allocation to the execution context.
@ kON_PROFILE_CHANGE
Reallocate for a profile when it's selected.
v_1_0::ILoggerFinder ILoggerFinder
Definition: NvInferRuntime.h:5315
LayerInformationFormat
The format in which the IEngineInspector prints the layer information.
Definition: NvInferRuntime.h:5041
@ kJSON
Print layer information in JSON format.
@ kONELINE
Print layer information in one line per layer.
v_1_0::IStreamReader IStreamReader
Definition: NvInferRuntime.h:700
AllocatorFlag
Allowed type of memory allocation.
Definition: NvInferRuntime.h:1553
@ kRESIZABLE
TensorRT may call realloc() on this allocation.
@ kMAX
Maximum over elements.
TensorLocation
The location for tensor data storage, device or host.
Definition: NvInferRuntime.h:214
@ kHOST
Data stored on host.
@ kDEVICE
Data stored on device.
OptProfileSelector
When setting or querying optimization profile parameters (such as shape tensor inputs or dynamic dime...
Definition: NvInferRuntime.h:2631
@ kOPT
This is used to set or get the value that is used in the optimization (kernel selection).
uint32_t AllocatorFlags
Definition: NvInferRuntime.h:1566
Severity
Enumerates severity levels for messages issued by the message recorder.
Definition: NvInferSafeRecorder.h:55
Summarizes tensors that a plugin might see for an input or output.
Definition: NvInferRuntime.h:373
Dims min
Lower bounds on tensor’s dimensions.
Definition: NvInferRuntime.h:378
Dims max
Upper bounds on tensor’s dimensions.
Definition: NvInferRuntime.h:381
Dims opt
Optimum value of tensor’s dimensions specified for auto-tuning.
Definition: NvInferRuntime.h:384
PluginTensorDesc desc
Information required to interpret a pointer to tensor data, except that desc.dims has -1 in place of ...
Definition: NvInferRuntime.h:375
Plugin field collection struct.
Definition: NvInferPluginBase.h:103
Fields that a plugin might see for an input or output.
Definition: NvInferRuntimePlugin.h:73
Declaration of EnumMaxImpl struct to store the exclusive upper bound of an enumeration type.
Definition: NvInferRuntimeBase.h:132

  Copyright © 2024 NVIDIA Corporation
  Privacy Policy | Manage My Privacy | Do Not Sell or Share My Data | Terms of Service | Accessibility | Corporate Policies | Product Security | Contact