This is an automated email from the ASF dual-hosted git repository.
tqchen pushed a commit to branch asf-site
in repository https://gitbox.apache.org/repos/asf/tvm-site.git
The following commit(s) were added to refs/heads/asf-site by this push:
new ea63b797e1 deploying docs
(apache/tvm@010089eafed221fd73e6f4fb9cd582ac3d63e2e7)
ea63b797e1 is described below
commit ea63b797e1d4ca3fe6f88fc8df96761d9d2cff33
Author: tvm-bot <[email protected]>
AuthorDate: Tue Sep 9 01:49:59 2025 +0000
deploying docs (apache/tvm@010089eafed221fd73e6f4fb9cd582ac3d63e2e7)
---
.../11c11e53c7dace51a8be968ee169ed0d/ir_module.zip | Bin 23904 -> 23904 bytes
.../tir_transformation.zip | Bin 15599 -> 15599 bytes
.../relax_creation.zip | Bin 22392 -> 22392 bytes
.../relax_transformation.zip | Bin 11460 -> 11460 bytes
.../optimize_llm.zip | Bin 54101 -> 54101 bytes
.../e2e_opt_model.zip | Bin 11237 -> 11237 bytes
.../quick_start.zip | Bin 16250 -> 16250 bytes
.../tir_creation.zip | Bin 24379 -> 24379 bytes
.../cross_compilation_and_rpc.zip | Bin 21700 -> 21700 bytes
.../customize_opt.zip | Bin 19813 -> 19813 bytes
.../relax/tutorials/sg_execution_times.rst.txt | 6 ++---
.../tensor_ir/tutorials/sg_execution_times.rst.txt | 6 ++---
.../tensor_ir/tutorials/tir_creation.rst.txt | 20 ++++++++--------
.../tensor_ir/tutorials/tir_transformation.rst.txt | 6 ++---
.../get_started/tutorials/ir_module.rst.txt | 8 +++----
.../get_started/tutorials/quick_start.rst.txt | 4 ++--
.../tutorials/sg_execution_times.rst.txt | 6 ++---
.../tutorials/cross_compilation_and_rpc.rst.txt | 2 +-
.../how_to/tutorials/customize_opt.rst.txt | 4 ++--
.../how_to/tutorials/e2e_opt_model.rst.txt | 2 +-
.../how_to/tutorials/sg_execution_times.rst.txt | 10 ++++----
docs/_sources/sg_execution_times.rst.txt | 26 ++++++++++-----------
docs/deep_dive/relax/tutorials/relax_creation.html | 16 +++++++++++--
.../relax/tutorials/relax_transformation.html | 15 ++++++++++--
.../relax/tutorials/sg_execution_times.html | 6 ++---
.../tensor_ir/tutorials/sg_execution_times.html | 6 ++---
.../tensor_ir/tutorials/tir_creation.html | 20 ++++++++--------
.../tensor_ir/tutorials/tir_transformation.html | 6 ++---
docs/get_started/tutorials/ir_module.html | 16 ++++++-------
docs/get_started/tutorials/quick_start.html | 24 +++++++++----------
docs/get_started/tutorials/sg_execution_times.html | 6 ++---
.../tutorials/cross_compilation_and_rpc.html | 2 +-
docs/how_to/tutorials/customize_opt.html | 8 +++----
docs/how_to/tutorials/e2e_opt_model.html | 8 +++----
docs/how_to/tutorials/optimize_llm.html | 10 ++++----
docs/how_to/tutorials/sg_execution_times.html | 10 ++++----
docs/objects.inv | Bin 19410 -> 19401 bytes
docs/reference/api/python/runtime/vm.html | 2 +-
docs/searchindex.js | 2 +-
docs/sg_execution_times.html | 26 ++++++++++-----------
40 files changed, 153 insertions(+), 130 deletions(-)
diff --git a/docs/_downloads/11c11e53c7dace51a8be968ee169ed0d/ir_module.zip
b/docs/_downloads/11c11e53c7dace51a8be968ee169ed0d/ir_module.zip
index 0ad8138700..369cb192ed 100644
Binary files a/docs/_downloads/11c11e53c7dace51a8be968ee169ed0d/ir_module.zip
and b/docs/_downloads/11c11e53c7dace51a8be968ee169ed0d/ir_module.zip differ
diff --git
a/docs/_downloads/18ba0d2ee8120824175aaef66bc9c9bf/tir_transformation.zip
b/docs/_downloads/18ba0d2ee8120824175aaef66bc9c9bf/tir_transformation.zip
index 3361600fc3..fe9512dc67 100644
Binary files
a/docs/_downloads/18ba0d2ee8120824175aaef66bc9c9bf/tir_transformation.zip and
b/docs/_downloads/18ba0d2ee8120824175aaef66bc9c9bf/tir_transformation.zip differ
diff --git
a/docs/_downloads/4753776bbe68e7c9ee4d19117973fc8b/relax_creation.zip
b/docs/_downloads/4753776bbe68e7c9ee4d19117973fc8b/relax_creation.zip
index 3abcee4755..44c5543d6b 100644
Binary files
a/docs/_downloads/4753776bbe68e7c9ee4d19117973fc8b/relax_creation.zip and
b/docs/_downloads/4753776bbe68e7c9ee4d19117973fc8b/relax_creation.zip differ
diff --git
a/docs/_downloads/7d201684dfa095a5ea48d98e9a2ef7ad/relax_transformation.zip
b/docs/_downloads/7d201684dfa095a5ea48d98e9a2ef7ad/relax_transformation.zip
index f842607aa0..2d7355d975 100644
Binary files
a/docs/_downloads/7d201684dfa095a5ea48d98e9a2ef7ad/relax_transformation.zip and
b/docs/_downloads/7d201684dfa095a5ea48d98e9a2ef7ad/relax_transformation.zip
differ
diff --git a/docs/_downloads/83e85f38cf16f1d926d06615fd54095c/optimize_llm.zip
b/docs/_downloads/83e85f38cf16f1d926d06615fd54095c/optimize_llm.zip
index 500b9f861c..1b02316f19 100644
Binary files
a/docs/_downloads/83e85f38cf16f1d926d06615fd54095c/optimize_llm.zip and
b/docs/_downloads/83e85f38cf16f1d926d06615fd54095c/optimize_llm.zip differ
diff --git a/docs/_downloads/a7dd7652b2ad50f82d7b739ce3645799/e2e_opt_model.zip
b/docs/_downloads/a7dd7652b2ad50f82d7b739ce3645799/e2e_opt_model.zip
index 4e90ec586c..231fae7cc4 100644
Binary files
a/docs/_downloads/a7dd7652b2ad50f82d7b739ce3645799/e2e_opt_model.zip and
b/docs/_downloads/a7dd7652b2ad50f82d7b739ce3645799/e2e_opt_model.zip differ
diff --git a/docs/_downloads/bb7db6678496193ed0c55d3b95fa6778/quick_start.zip
b/docs/_downloads/bb7db6678496193ed0c55d3b95fa6778/quick_start.zip
index 7493414a4b..b36a8ad5fe 100644
Binary files a/docs/_downloads/bb7db6678496193ed0c55d3b95fa6778/quick_start.zip
and b/docs/_downloads/bb7db6678496193ed0c55d3b95fa6778/quick_start.zip differ
diff --git a/docs/_downloads/be26483bb70b8468499a01c55e8e866c/tir_creation.zip
b/docs/_downloads/be26483bb70b8468499a01c55e8e866c/tir_creation.zip
index c25ac63d8b..4de560268f 100644
Binary files
a/docs/_downloads/be26483bb70b8468499a01c55e8e866c/tir_creation.zip and
b/docs/_downloads/be26483bb70b8468499a01c55e8e866c/tir_creation.zip differ
diff --git
a/docs/_downloads/f69380821f417ef2210f45503d81bded/cross_compilation_and_rpc.zip
b/docs/_downloads/f69380821f417ef2210f45503d81bded/cross_compilation_and_rpc.zip
index 6c48a40ebd..9703be68f7 100644
Binary files
a/docs/_downloads/f69380821f417ef2210f45503d81bded/cross_compilation_and_rpc.zip
and
b/docs/_downloads/f69380821f417ef2210f45503d81bded/cross_compilation_and_rpc.zip
differ
diff --git a/docs/_downloads/f69433a4a80715725df90d1386679956/customize_opt.zip
b/docs/_downloads/f69433a4a80715725df90d1386679956/customize_opt.zip
index bd3944e067..64a34bb44c 100644
Binary files
a/docs/_downloads/f69433a4a80715725df90d1386679956/customize_opt.zip and
b/docs/_downloads/f69433a4a80715725df90d1386679956/customize_opt.zip differ
diff --git a/docs/_sources/deep_dive/relax/tutorials/sg_execution_times.rst.txt
b/docs/_sources/deep_dive/relax/tutorials/sg_execution_times.rst.txt
index f23da81783..83fb799c7f 100644
--- a/docs/_sources/deep_dive/relax/tutorials/sg_execution_times.rst.txt
+++ b/docs/_sources/deep_dive/relax/tutorials/sg_execution_times.rst.txt
@@ -6,7 +6,7 @@
Computation times
=================
-**00:00.207** total execution time for 2 files **from
deep_dive/relax/tutorials**:
+**00:00.204** total execution time for 2 files **from
deep_dive/relax/tutorials**:
.. container::
@@ -33,8 +33,8 @@ Computation times
- Time
- Mem (MB)
* - :ref:`sphx_glr_deep_dive_relax_tutorials_relax_creation.py`
(``relax_creation.py``)
- - 00:00.131
+ - 00:00.127
- 0.0
* - :ref:`sphx_glr_deep_dive_relax_tutorials_relax_transformation.py`
(``relax_transformation.py``)
- - 00:00.075
+ - 00:00.077
- 0.0
diff --git
a/docs/_sources/deep_dive/tensor_ir/tutorials/sg_execution_times.rst.txt
b/docs/_sources/deep_dive/tensor_ir/tutorials/sg_execution_times.rst.txt
index 56944ba78d..2bd6644c17 100644
--- a/docs/_sources/deep_dive/tensor_ir/tutorials/sg_execution_times.rst.txt
+++ b/docs/_sources/deep_dive/tensor_ir/tutorials/sg_execution_times.rst.txt
@@ -6,7 +6,7 @@
Computation times
=================
-**00:00.501** total execution time for 2 files **from
deep_dive/tensor_ir/tutorials**:
+**00:00.497** total execution time for 2 files **from
deep_dive/tensor_ir/tutorials**:
.. container::
@@ -33,8 +33,8 @@ Computation times
- Time
- Mem (MB)
* - :ref:`sphx_glr_deep_dive_tensor_ir_tutorials_tir_transformation.py`
(``tir_transformation.py``)
- - 00:00.312
+ - 00:00.309
- 0.0
* - :ref:`sphx_glr_deep_dive_tensor_ir_tutorials_tir_creation.py`
(``tir_creation.py``)
- - 00:00.189
+ - 00:00.188
- 0.0
diff --git a/docs/_sources/deep_dive/tensor_ir/tutorials/tir_creation.rst.txt
b/docs/_sources/deep_dive/tensor_ir/tutorials/tir_creation.rst.txt
index 663e96b031..e8c4d48bdc 100644
--- a/docs/_sources/deep_dive/tensor_ir/tutorials/tir_creation.rst.txt
+++ b/docs/_sources/deep_dive/tensor_ir/tutorials/tir_creation.rst.txt
@@ -325,17 +325,17 @@ Now let's check the runtime dynamic shape inference:
.. code-block:: none
- [[1.5833187 1.2412384 1.0804644 1.1017636 ]
- [1.2663358 0.75536585 0.8446858 0.8348684 ]
- [1.6010246 0.7777941 0.9558423 1.2165275 ]
- [0.95500666 1.0444615 0.78498745 0.68005484]]
- [[33.429916 27.969305 30.368662 ... 28.219225 29.495378 32.23988 ]
- [34.71739 32.470253 32.717396 ... 30.84613 29.42731 35.19347 ]
- [34.89939 31.082674 33.39309 ... 32.72572 32.869007 35.204777]
+ [[1.1267889 1.4921708 1.4765959 0.9249995]
+ [2.2765992 2.1712847 2.6449652 1.5365322]
+ [2.0343807 1.4992819 2.2396464 1.2175505]
+ [1.5460057 1.5706404 1.744776 1.0376716]]
+ [[33.281487 31.678526 32.2581 ... 31.762579 36.60329 36.011604]
+ [30.123138 28.297188 29.13717 ... 28.510406 32.734715 29.437414]
+ [30.358988 26.641193 27.951345 ... 29.039057 30.69999 30.044214]
...
- [35.31894 31.633984 33.407932 ... 32.39267 31.336895 35.725048]
- [35.153107 31.38942 30.087236 ... 32.814247 29.592163 34.438763]
- [34.176086 31.727692 32.234245 ... 31.901295 30.515553 34.1984 ]]
+ [33.876442 30.334793 32.056587 ... 31.329277 33.20621 32.426105]
+ [32.626663 30.220213 31.663929 ... 31.735876 34.201866 33.056576]
+ [33.54827 32.28254 30.95808 ... 29.824718 34.815678 32.64949 ]]
diff --git
a/docs/_sources/deep_dive/tensor_ir/tutorials/tir_transformation.rst.txt
b/docs/_sources/deep_dive/tensor_ir/tutorials/tir_transformation.rst.txt
index 096d369960..86b74b0621 100644
--- a/docs/_sources/deep_dive/tensor_ir/tutorials/tir_transformation.rst.txt
+++ b/docs/_sources/deep_dive/tensor_ir/tutorials/tir_transformation.rst.txt
@@ -123,7 +123,7 @@ original implementation.
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 2.2302 2.2302 2.2302 2.2302 0.0000
+ 2.2319 2.2319 2.2319 2.2319 0.0000
@@ -295,7 +295,7 @@ action involves reordering these two loops.
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 0.8467 0.8467 0.8467 0.8467 0.0000
+ 0.8472 0.8472 0.8472 0.8472 0.0000
@@ -423,7 +423,7 @@ from the reduction update via the **decompose_reduction**
primitive.
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 0.3178 0.3178 0.3178 0.3178 0.0000
+ 0.3191 0.3191 0.3191 0.3191 0.0000
diff --git a/docs/_sources/get_started/tutorials/ir_module.rst.txt
b/docs/_sources/get_started/tutorials/ir_module.rst.txt
index 635281c6a7..4397066557 100644
--- a/docs/_sources/get_started/tutorials/ir_module.rst.txt
+++ b/docs/_sources/get_started/tutorials/ir_module.rst.txt
@@ -698,8 +698,8 @@ We can deploy the IRModule on CPU by specifying the target
as ``llvm``.
.. code-block:: none
- [[-0.10236609 -0.16140339 -0.13421026 -0.13498658 -0.01205717 -0.0592057
- 0.06018345 -0.23712799 -0.03988083 -0.11772955]]
+ [[-0.00110031 -0.20449686 -0.14811155 0.13757232 -0.20253748 0.04177996
+ -0.07720898 0.14182252 -0.11631804 0.08094296]]
@@ -765,8 +765,8 @@ Now we can compile the IRModule on GPU, the similar way as
we did on CPU.
.. code-block:: none
- [[-0.10236608 -0.16140339 -0.13421026 -0.13498661 -0.01205715 -0.05920573
- 0.06018349 -0.23712796 -0.03988081 -0.11772955]]
+ [[-0.00110032 -0.20449692 -0.1481115 0.13757232 -0.20253742 0.04177994
+ -0.07720903 0.14182247 -0.11631814 0.08094295]]
diff --git a/docs/_sources/get_started/tutorials/quick_start.rst.txt
b/docs/_sources/get_started/tutorials/quick_start.rst.txt
index a67fc4229d..302cfbb031 100644
--- a/docs/_sources/get_started/tutorials/quick_start.rst.txt
+++ b/docs/_sources/get_started/tutorials/quick_start.rst.txt
@@ -230,8 +230,8 @@ different devices.
.. code-block:: none
- [[26480.525 26398.688 24851.756 26216.059 25056.197 26085.629 25934.707
- 25167.797 26442.43 25794.87 ]]
+ [[23709.96 23185.262 22688.762 24297.123 22736.96 23683.799 23711.588
+ 24120.336 23535.51 24328.434]]
diff --git a/docs/_sources/get_started/tutorials/sg_execution_times.rst.txt
b/docs/_sources/get_started/tutorials/sg_execution_times.rst.txt
index 897cfb1720..092da15be0 100644
--- a/docs/_sources/get_started/tutorials/sg_execution_times.rst.txt
+++ b/docs/_sources/get_started/tutorials/sg_execution_times.rst.txt
@@ -6,7 +6,7 @@
Computation times
=================
-**00:10.678** total execution time for 2 files **from get_started/tutorials**:
+**00:10.212** total execution time for 2 files **from get_started/tutorials**:
.. container::
@@ -33,8 +33,8 @@ Computation times
- Time
- Mem (MB)
* - :ref:`sphx_glr_get_started_tutorials_ir_module.py` (``ir_module.py``)
- - 00:10.477
+ - 00:10.028
- 0.0
* - :ref:`sphx_glr_get_started_tutorials_quick_start.py`
(``quick_start.py``)
- - 00:00.201
+ - 00:00.184
- 0.0
diff --git a/docs/_sources/how_to/tutorials/cross_compilation_and_rpc.rst.txt
b/docs/_sources/how_to/tutorials/cross_compilation_and_rpc.rst.txt
index cbd5ad4a73..8394d20a52 100644
--- a/docs/_sources/how_to/tutorials/cross_compilation_and_rpc.rst.txt
+++ b/docs/_sources/how_to/tutorials/cross_compilation_and_rpc.rst.txt
@@ -274,7 +274,7 @@ device and returns the measured cost. Network overhead is
excluded.
.. code-block:: none
- 1.319e-07 secs/op
+ 1.2781e-06 secs/op
diff --git a/docs/_sources/how_to/tutorials/customize_opt.rst.txt
b/docs/_sources/how_to/tutorials/customize_opt.rst.txt
index a97ab863c1..6782669b83 100644
--- a/docs/_sources/how_to/tutorials/customize_opt.rst.txt
+++ b/docs/_sources/how_to/tutorials/customize_opt.rst.txt
@@ -420,8 +420,8 @@ We can build and deploy the optimized model to the TVM
runtime.
.. code-block:: none
- [[23666.377 26352.043 24360.344 25105.71 25748.316 24053.691 25034.05
- 26084.346 24839.422 26120.78 ]]
+ [[25804.072 25234.797 24747.855 23863.885 24310.018 26154.75 25283.268
+ 24277.625 25507.771 25346.963]]
diff --git a/docs/_sources/how_to/tutorials/e2e_opt_model.rst.txt
b/docs/_sources/how_to/tutorials/e2e_opt_model.rst.txt
index b43ebae558..4135b3bdf7 100644
--- a/docs/_sources/how_to/tutorials/e2e_opt_model.rst.txt
+++ b/docs/_sources/how_to/tutorials/e2e_opt_model.rst.txt
@@ -59,7 +59,7 @@ PyTorch.
.. code-block:: none
Downloading: "https://download.pytorch.org/models/resnet18-f37072fd.pth"
to /workspace/.cache/torch/hub/checkpoints/resnet18-f37072fd.pth
- 0%| | 0.00/44.7M [00:00<?, ?B/s] 40%|████ |
18.0M/44.7M [00:00<00:00, 188MB/s] 92%|█████████▏| 41.0M/44.7M
[00:00<00:00, 219MB/s] 100%|██████████| 44.7M/44.7M [00:00<00:00, 216MB/s]
+ 0%| | 0.00/44.7M [00:00<?, ?B/s] 36%|███▌ |
16.1M/44.7M [00:00<00:00, 169MB/s] 86%|████████▌ | 38.4M/44.7M
[00:00<00:00, 207MB/s] 100%|██████████| 44.7M/44.7M [00:00<00:00, 205MB/s]
diff --git a/docs/_sources/how_to/tutorials/sg_execution_times.rst.txt
b/docs/_sources/how_to/tutorials/sg_execution_times.rst.txt
index cae647c782..9c24349b62 100644
--- a/docs/_sources/how_to/tutorials/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/tutorials/sg_execution_times.rst.txt
@@ -6,7 +6,7 @@
Computation times
=================
-**00:40.736** total execution time for 4 files **from how_to/tutorials**:
+**00:39.081** total execution time for 4 files **from how_to/tutorials**:
.. container::
@@ -33,14 +33,14 @@ Computation times
- Time
- Mem (MB)
* - :ref:`sphx_glr_how_to_tutorials_optimize_llm.py` (``optimize_llm.py``)
- - 00:39.050
+ - 00:37.439
- 0.0
* - :ref:`sphx_glr_how_to_tutorials_e2e_opt_model.py` (``e2e_opt_model.py``)
- - 00:00.768
+ - 00:00.739
- 0.0
* - :ref:`sphx_glr_how_to_tutorials_customize_opt.py` (``customize_opt.py``)
- - 00:00.708
+ - 00:00.666
- 0.0
* - :ref:`sphx_glr_how_to_tutorials_cross_compilation_and_rpc.py`
(``cross_compilation_and_rpc.py``)
- - 00:00.210
+ - 00:00.237
- 0.0
diff --git a/docs/_sources/sg_execution_times.rst.txt
b/docs/_sources/sg_execution_times.rst.txt
index 0d280a515e..fc9822f5d9 100644
--- a/docs/_sources/sg_execution_times.rst.txt
+++ b/docs/_sources/sg_execution_times.rst.txt
@@ -6,7 +6,7 @@
Computation times
=================
-**00:52.122** total execution time for 10 files **from all galleries**:
+**00:49.994** total execution time for 10 files **from all galleries**:
.. container::
@@ -33,32 +33,32 @@ Computation times
- Time
- Mem (MB)
* - :ref:`sphx_glr_how_to_tutorials_optimize_llm.py`
(``../how_to/tutorials/optimize_llm.py``)
- - 00:39.050
+ - 00:37.439
- 0.0
* - :ref:`sphx_glr_get_started_tutorials_ir_module.py`
(``../get_started/tutorials/ir_module.py``)
- - 00:10.477
+ - 00:10.028
- 0.0
* - :ref:`sphx_glr_how_to_tutorials_e2e_opt_model.py`
(``../how_to/tutorials/e2e_opt_model.py``)
- - 00:00.768
+ - 00:00.739
- 0.0
* - :ref:`sphx_glr_how_to_tutorials_customize_opt.py`
(``../how_to/tutorials/customize_opt.py``)
- - 00:00.708
+ - 00:00.666
- 0.0
* - :ref:`sphx_glr_deep_dive_tensor_ir_tutorials_tir_transformation.py`
(``../deep_dive/tensor_ir/tutorials/tir_transformation.py``)
- - 00:00.312
+ - 00:00.309
- 0.0
* - :ref:`sphx_glr_how_to_tutorials_cross_compilation_and_rpc.py`
(``../how_to/tutorials/cross_compilation_and_rpc.py``)
- - 00:00.210
- - 0.0
- * - :ref:`sphx_glr_get_started_tutorials_quick_start.py`
(``../get_started/tutorials/quick_start.py``)
- - 00:00.201
+ - 00:00.237
- 0.0
* - :ref:`sphx_glr_deep_dive_tensor_ir_tutorials_tir_creation.py`
(``../deep_dive/tensor_ir/tutorials/tir_creation.py``)
- - 00:00.189
+ - 00:00.188
+ - 0.0
+ * - :ref:`sphx_glr_get_started_tutorials_quick_start.py`
(``../get_started/tutorials/quick_start.py``)
+ - 00:00.184
- 0.0
* - :ref:`sphx_glr_deep_dive_relax_tutorials_relax_creation.py`
(``../deep_dive/relax/tutorials/relax_creation.py``)
- - 00:00.131
+ - 00:00.127
- 0.0
* - :ref:`sphx_glr_deep_dive_relax_tutorials_relax_transformation.py`
(``../deep_dive/relax/tutorials/relax_transformation.py``)
- - 00:00.075
+ - 00:00.077
- 0.0
diff --git a/docs/deep_dive/relax/tutorials/relax_creation.html
b/docs/deep_dive/relax/tutorials/relax_creation.html
index 049af91697..480e7628f9 100644
--- a/docs/deep_dive/relax/tutorials/relax_creation.html
+++ b/docs/deep_dive/relax/tutorials/relax_creation.html
@@ -191,10 +191,22 @@
<li class="toctree-l1"><a class="reference internal"
href="../../../how_to/dev/index.html">Development Guides</a></li>
</ul>
<p class="caption" role="heading"><span class="caption-text">Deep
Dive</span></p>
-<ul>
+<ul class="current">
<li class="toctree-l1"><a class="reference internal"
href="../../../arch/index.html">Design and Architecture</a></li>
<li class="toctree-l1"><a class="reference internal"
href="../../tensor_ir/index.html">TensorIR</a></li>
-<li class="toctree-l1"><a class="reference internal"
href="../index.html">Relax</a></li>
+<li class="toctree-l1 current"><a class="reference internal"
href="../index.html">Relax</a><ul class="current">
+<li class="toctree-l2"><a class="reference internal"
href="../abstraction.html">Graph Abstraction for ML Models</a></li>
+<li class="toctree-l2"><a class="reference internal"
href="../learning.html">Understand Relax Abstraction</a></li>
+<li class="toctree-l2 current"><a class="current reference internal"
href="#">Relax Creation</a><ul>
+<li class="toctree-l3"><a class="reference internal"
href="#create-relax-programs-using-tvmscript">Create Relax programs using
TVMScript</a></li>
+<li class="toctree-l3"><a class="reference internal"
href="#create-relax-programs-using-nnmodule-api">Create Relax programs using
NNModule API</a></li>
+<li class="toctree-l3"><a class="reference internal"
href="#create-relax-programs-using-block-builder-api">Create Relax programs
using Block Builder API</a></li>
+<li class="toctree-l3"><a class="reference internal"
href="#summary">Summary</a></li>
+</ul>
+</li>
+<li class="toctree-l2"><a class="reference internal"
href="relax_transformation.html">Transformation</a></li>
+</ul>
+</li>
</ul>
<p class="caption" role="heading"><span class="caption-text">API
Reference</span></p>
<ul>
diff --git a/docs/deep_dive/relax/tutorials/relax_transformation.html
b/docs/deep_dive/relax/tutorials/relax_transformation.html
index a6c7e8bf6b..e6c76a0897 100644
--- a/docs/deep_dive/relax/tutorials/relax_transformation.html
+++ b/docs/deep_dive/relax/tutorials/relax_transformation.html
@@ -191,10 +191,21 @@
<li class="toctree-l1"><a class="reference internal"
href="../../../how_to/dev/index.html">Development Guides</a></li>
</ul>
<p class="caption" role="heading"><span class="caption-text">Deep
Dive</span></p>
-<ul>
+<ul class="current">
<li class="toctree-l1"><a class="reference internal"
href="../../../arch/index.html">Design and Architecture</a></li>
<li class="toctree-l1"><a class="reference internal"
href="../../tensor_ir/index.html">TensorIR</a></li>
-<li class="toctree-l1"><a class="reference internal"
href="../index.html">Relax</a></li>
+<li class="toctree-l1 current"><a class="reference internal"
href="../index.html">Relax</a><ul class="current">
+<li class="toctree-l2"><a class="reference internal"
href="../abstraction.html">Graph Abstraction for ML Models</a></li>
+<li class="toctree-l2"><a class="reference internal"
href="../learning.html">Understand Relax Abstraction</a></li>
+<li class="toctree-l2"><a class="reference internal"
href="relax_creation.html">Relax Creation</a></li>
+<li class="toctree-l2 current"><a class="current reference internal"
href="#">Transformation</a><ul>
+<li class="toctree-l3"><a class="reference internal"
href="#apply-transformations">Apply transformations</a></li>
+<li class="toctree-l3"><a class="reference internal"
href="#custom-passes">Custom Passes</a></li>
+<li class="toctree-l3"><a class="reference internal"
href="#summary">Summary</a></li>
+</ul>
+</li>
+</ul>
+</li>
</ul>
<p class="caption" role="heading"><span class="caption-text">API
Reference</span></p>
<ul>
diff --git a/docs/deep_dive/relax/tutorials/sg_execution_times.html
b/docs/deep_dive/relax/tutorials/sg_execution_times.html
index da75ed4974..5c5253ae81 100644
--- a/docs/deep_dive/relax/tutorials/sg_execution_times.html
+++ b/docs/deep_dive/relax/tutorials/sg_execution_times.html
@@ -293,7 +293,7 @@
<section id="computation-times">
<span
id="sphx-glr-deep-dive-relax-tutorials-sg-execution-times"></span><h1>Computation
times<a class="headerlink" href="#computation-times" title="Link to this
heading"></a></h1>
-<p><strong>00:00.207</strong> total execution time for 2 files <strong>from
deep_dive/relax/tutorials</strong>:</p>
+<p><strong>00:00.204</strong> total execution time for 2 files <strong>from
deep_dive/relax/tutorials</strong>:</p>
<div class="docutils container">
<style scoped>
<link
href="https://cdnjs.cloudflare.com/ajax/libs/twitter-bootstrap/5.3.0/css/bootstrap.min.css"
rel="stylesheet" />
@@ -315,11 +315,11 @@ $(document).ready( function () {
</thead>
<tbody>
<tr class="row-even"><td><p><a class="reference internal"
href="relax_creation.html#sphx-glr-deep-dive-relax-tutorials-relax-creation-py"><span
class="std std-ref">Relax Creation</span></a> (<code class="docutils literal
notranslate"><span class="pre">relax_creation.py</span></code>)</p></td>
-<td><p>00:00.131</p></td>
+<td><p>00:00.127</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="relax_transformation.html#sphx-glr-deep-dive-relax-tutorials-relax-transformation-py"><span
class="std std-ref">Transformation</span></a> (<code class="docutils literal
notranslate"><span class="pre">relax_transformation.py</span></code>)</p></td>
-<td><p>00:00.075</p></td>
+<td><p>00:00.077</p></td>
<td><p>0.0</p></td>
</tr>
</tbody>
diff --git a/docs/deep_dive/tensor_ir/tutorials/sg_execution_times.html
b/docs/deep_dive/tensor_ir/tutorials/sg_execution_times.html
index 658030ce21..9d33f9a096 100644
--- a/docs/deep_dive/tensor_ir/tutorials/sg_execution_times.html
+++ b/docs/deep_dive/tensor_ir/tutorials/sg_execution_times.html
@@ -293,7 +293,7 @@
<section id="computation-times">
<span
id="sphx-glr-deep-dive-tensor-ir-tutorials-sg-execution-times"></span><h1>Computation
times<a class="headerlink" href="#computation-times" title="Link to this
heading"></a></h1>
-<p><strong>00:00.501</strong> total execution time for 2 files <strong>from
deep_dive/tensor_ir/tutorials</strong>:</p>
+<p><strong>00:00.497</strong> total execution time for 2 files <strong>from
deep_dive/tensor_ir/tutorials</strong>:</p>
<div class="docutils container">
<style scoped>
<link
href="https://cdnjs.cloudflare.com/ajax/libs/twitter-bootstrap/5.3.0/css/bootstrap.min.css"
rel="stylesheet" />
@@ -315,11 +315,11 @@ $(document).ready( function () {
</thead>
<tbody>
<tr class="row-even"><td><p><a class="reference internal"
href="tir_transformation.html#sphx-glr-deep-dive-tensor-ir-tutorials-tir-transformation-py"><span
class="std std-ref">Transformation</span></a> (<code class="docutils literal
notranslate"><span class="pre">tir_transformation.py</span></code>)</p></td>
-<td><p>00:00.312</p></td>
+<td><p>00:00.309</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="tir_creation.html#sphx-glr-deep-dive-tensor-ir-tutorials-tir-creation-py"><span
class="std std-ref">TensorIR Creation</span></a> (<code class="docutils
literal notranslate"><span class="pre">tir_creation.py</span></code>)</p></td>
-<td><p>00:00.189</p></td>
+<td><p>00:00.188</p></td>
<td><p>0.0</p></td>
</tr>
</tbody>
diff --git a/docs/deep_dive/tensor_ir/tutorials/tir_creation.html
b/docs/deep_dive/tensor_ir/tutorials/tir_creation.html
index fa5f63b120..8d3e54df51 100644
--- a/docs/deep_dive/tensor_ir/tutorials/tir_creation.html
+++ b/docs/deep_dive/tensor_ir/tutorials/tir_creation.html
@@ -491,17 +491,17 @@ be used to ascertain the shape and data type of a
TensorIR.</p>
<span class="nb">print</span><span class="p">(</span><span
class="n">evaluate_dynamic_shape</span><span class="p">(</span><span
class="n">dyn_shape_lib</span><span class="p">,</span> <span
class="n">m</span><span class="o">=</span><span class="mi">64</span><span
class="p">,</span> <span class="n">n</span><span class="o">=</span><span
class="mi">64</span><span class="p">,</span> <a
href="../../../reference/api/python/tir/tir.html#tvm.tir.IterVar"
title="tvm.tir.IterVar" class="sphx-glr-ba [...]
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[1.5833187 1.2412384 1.0804644
1.1017636 ]
- [1.2663358 0.75536585 0.8446858 0.8348684 ]
- [1.6010246 0.7777941 0.9558423 1.2165275 ]
- [0.95500666 1.0444615 0.78498745 0.68005484]]
-[[33.429916 27.969305 30.368662 ... 28.219225 29.495378 32.23988 ]
- [34.71739 32.470253 32.717396 ... 30.84613 29.42731 35.19347 ]
- [34.89939 31.082674 33.39309 ... 32.72572 32.869007 35.204777]
+<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[1.1267889 1.4921708 1.4765959 0.9249995]
+ [2.2765992 2.1712847 2.6449652 1.5365322]
+ [2.0343807 1.4992819 2.2396464 1.2175505]
+ [1.5460057 1.5706404 1.744776 1.0376716]]
+[[33.281487 31.678526 32.2581 ... 31.762579 36.60329 36.011604]
+ [30.123138 28.297188 29.13717 ... 28.510406 32.734715 29.437414]
+ [30.358988 26.641193 27.951345 ... 29.039057 30.69999 30.044214]
...
- [35.31894 31.633984 33.407932 ... 32.39267 31.336895 35.725048]
- [35.153107 31.38942 30.087236 ... 32.814247 29.592163 34.438763]
- [34.176086 31.727692 32.234245 ... 31.901295 30.515553 34.1984 ]]
+ [33.876442 30.334793 32.056587 ... 31.329277 33.20621 32.426105]
+ [32.626663 30.220213 31.663929 ... 31.735876 34.201866 33.056576]
+ [33.54827 32.28254 30.95808 ... 29.824718 34.815678 32.64949 ]]
</pre></div>
</div>
</section>
diff --git a/docs/deep_dive/tensor_ir/tutorials/tir_transformation.html
b/docs/deep_dive/tensor_ir/tutorials/tir_transformation.html
index 80ee84c9b6..f20fb25646 100644
--- a/docs/deep_dive/tensor_ir/tutorials/tir_transformation.html
+++ b/docs/deep_dive/tensor_ir/tutorials/tir_transformation.html
@@ -370,7 +370,7 @@ original implementation.</p>
</div>
<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 2.2302 2.2302 2.2302 2.2302 0.0000
+ 2.2319 2.2319 2.2319 2.2319 0.0000
</pre></div>
</div>
<section id="initialization-schedule">
@@ -466,7 +466,7 @@ class Module:
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 0.8467 0.8467 0.8467 0.8467 0.0000
+ 0.8472 0.8472 0.8472 0.8472 0.0000
</pre></div>
</div>
</section>
@@ -560,7 +560,7 @@ class Module:
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 0.3178 0.3178 0.3178 0.3178 0.0000
+ 0.3191 0.3191 0.3191 0.3191 0.0000
</pre></div>
</div>
</section>
diff --git a/docs/get_started/tutorials/ir_module.html
b/docs/get_started/tutorials/ir_module.html
index fff604082a..e96a7bdf41 100644
--- a/docs/get_started/tutorials/ir_module.html
+++ b/docs/get_started/tutorials/ir_module.html
@@ -805,16 +805,16 @@ backends.</p>
<p>We can deploy the IRModule on CPU by specifying the target as <code
class="docutils literal notranslate"><span class="pre">llvm</span></code>.</p>
<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">exec</span></a> <span class="o">=</span> <a
href="../../reference/api/python/driver.html#tvm.compile" title="tvm.compile"
class="sphx-glr-backref-module-tvm sphx-glr-backref-type-py-func [...]
<span class="n">dev</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">cpu</span><span
class="p">()</span>
-<span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/relax/relax.html#tvm.relax.VirtualMachine"
title="tvm.relax.VirtualMachine" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">relax</span><span class="o">.</span><span
class="n">VirtualMachine</span></a><span class="p">(</span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class [...]
+<a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a> <span
class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm
sphx-glr-backref-type-py-class"><span class=" [...]
<span class="n">raw_data</span> <span class="o">=</span> <span
class="n">np</span><span class="o">.</span><span class="n">random</span><span
class="o">.</span><span class="n">rand</span><span class="p">(</span><span
class="mi">1</span><span class="p">,</span> <span class="mi">784</span><span
class="p">)</span><span class="o">.</span><span class="n">astype</span><span
class="p">(</span><span class="s2">"float32"</span><span
class="p">)</span>
<span class="n">data</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">runtime</span><span
class="o">.</span><span class="n">tensor</span><span class="p">(</span><span
class="n">raw_data</span><span class="p">,</span> <span
class="n">dev</span><span class="p">)</span>
-<span class="n">cpu_out</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"main"</span><span class="p">](</span><span
class="n">data</span><span class="p">,</span> <span class="o">*</span><a
href="https://docs.python.org/3/library/stdtypes.html#dict"
title="builtins.dict" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">params_from_torch</span></a><span class="p">[</ [...]
+<span class="n">cpu_out</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"main"</span><span
class="p">](</span><span class="n">data</span><span class="p">,</span> <span
class="o">*</span><a href="https:// [...]
<span class="nb">print</span><span class="p">(</span><span
class="n">cpu_out</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[-0.10236609 -0.16140339 -0.13421026
-0.13498658 -0.01205717 -0.0592057
- 0.06018345 -0.23712799 -0.03988083 -0.11772955]]
+<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[-0.00110031 -0.20449686 -0.14811155
0.13757232 -0.20253748 0.04177996
+ -0.07720898 0.14182252 -0.11631804 0.08094296]]
</pre></div>
</div>
</section>
@@ -837,19 +837,19 @@ the details of <code class="docutils literal
notranslate"><span class="pre">DLig
<p>Now we can compile the IRModule on GPU, the similar way as we did on
CPU.</p>
<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">exec</span></a> <span class="o">=</span> <a
href="../../reference/api/python/driver.html#tvm.compile" title="tvm.compile"
class="sphx-glr-backref-module-tvm sphx-glr-backref-type-py-func [...]
<span class="n">dev</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">device</span><span
class="p">(</span><span class="s2">"cuda"</span><span
class="p">,</span> <span class="mi">0</span><span class="p">)</span>
-<span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/relax/relax.html#tvm.relax.VirtualMachine"
title="tvm.relax.VirtualMachine" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">relax</span><span class="o">.</span><span
class="n">VirtualMachine</span></a><span class="p">(</span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class [...]
+<a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a> <span
class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm
sphx-glr-backref-type-py-class"><span class=" [...]
<span class="c1"># Need to allocate data and params on GPU device</span>
<span class="n">data</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">runtime</span><span
class="o">.</span><span class="n">tensor</span><span class="p">(</span><span
class="n">raw_data</span><span class="p">,</span> <span
class="n">dev</span><span class="p">)</span>
<a href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">gpu_params</span></a> <span class="o">=</span> <span
class="p">[</span><span class="n">tvm</span><span class="o">.</span><span
class="n">runtime</span><span class="o">.</span><span
class="n">tensor</span><span class="p">(</span><span class="n">p</span><span
class="p">,</span> <span class="n"> [...]
-<span class="n">gpu_out</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"main"</span><span class="p">](</span><span
class="n">data</span><span class="p">,</span> <span class="o">*</span><a
href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">gpu_params</span></a><span class="p">)</span><s [...]
+<span class="n">gpu_out</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"main"</span><span
class="p">](</span><span class="n">data</span><span class="p">,</span> <span
class="o">*</span><a href="https:// [...]
<span class="nb">print</span><span class="p">(</span><span
class="n">gpu_out</span><span class="p">)</span>
<span class="c1"># Check the correctness of the results</span>
<span class="k">assert</span> <span class="n">np</span><span
class="o">.</span><span class="n">allclose</span><span class="p">(</span><span
class="n">cpu_out</span><span class="p">,</span> <span
class="n">gpu_out</span><span class="p">,</span> <span
class="n">atol</span><span class="o">=</span><span class="mf">1e-3</span><span
class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[-0.10236608 -0.16140339 -0.13421026
-0.13498661 -0.01205715 -0.05920573
- 0.06018349 -0.23712796 -0.03988081 -0.11772955]]
+<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[-0.00110032 -0.20449692 -0.1481115
0.13757232 -0.20253742 0.04177994
+ -0.07720903 0.14182247 -0.11631814 0.08094295]]
</pre></div>
</div>
</section>
diff --git a/docs/get_started/tutorials/quick_start.html
b/docs/get_started/tutorials/quick_start.html
index 05c322131c..dfd5281f30 100644
--- a/docs/get_started/tutorials/quick_start.html
+++ b/docs/get_started/tutorials/quick_start.html
@@ -451,16 +451,16 @@ different devices.</p>
<a href="../../reference/api/python/target.html#tvm.target.Target"
title="tvm.target.Target" class="sphx-glr-backref-module-tvm-target
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">target</span></a> <span class="o">=</span> <a
href="../../reference/api/python/target.html#tvm.target.Target"
title="tvm.target.Target" class="sphx-glr-backref-module-tvm-target
sphx-glr-backref-type-py-class"><span class="n">tvm</span><span
class="o">.</span><span class="n">target< [...]
<a href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">ex</span></a> <span class="o">=</span> <a
href="../../reference/api/python/driver.html#tvm.compile" title="tvm.compile"
class="sphx-glr-backref-module-tvm sphx-glr-backref-type-py-function"><span
class="n">tvm</span><span class="o">.</span><span class="n">compile</span [...]
<span class="n">device</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">cpu</span><span
class="p">()</span>
-<span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/relax/relax.html#tvm.relax.VirtualMachine"
title="tvm.relax.VirtualMachine" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">relax</span><span class="o">.</span><span
class="n">VirtualMachine</span></a><span class="p">(</span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class [...]
+<a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a> <span
class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm
sphx-glr-backref-type-py-class"><span class=" [...]
<span class="n">data</span> <span class="o">=</span> <span
class="n">np</span><span class="o">.</span><span class="n">random</span><span
class="o">.</span><span class="n">rand</span><span class="p">(</span><span
class="mi">1</span><span class="p">,</span> <span class="mi">784</span><span
class="p">)</span><span class="o">.</span><span class="n">astype</span><span
class="p">(</span><span class="s2">"float32"</span><span
class="p">)</span>
<span class="n">tvm_data</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">runtime</span><span
class="o">.</span><span class="n">tensor</span><span class="p">(</span><span
class="n">data</span><span class="p">,</span> <span
class="n">device</span><span class="o">=</span><span
class="n">device</span><span class="p">)</span>
<a href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">params</span></a> <span class="o">=</span> <span
class="p">[</span><span class="n">np</span><span class="o">.</span><span
class="n">random</span><span class="o">.</span><span class="n">rand</span><span
class="p">(</span><span class="o">*</span><span class="n">param</span><span
class="o">.</sp [...]
<a href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">params</span></a> <span class="o">=</span> <span
class="p">[</span><span class="n">tvm</span><span class="o">.</span><span
class="n">runtime</span><span class="o">.</span><span
class="n">tensor</span><span class="p">(</span><span
class="n">param</span><span class="p">,</span> <span class="n"> [...]
-<span class="nb">print</span><span class="p">(</span><span
class="n">vm</span><span class="p">[</span><span
class="s2">"forward"</span><span class="p">](</span><span
class="n">tvm_data</span><span class="p">,</span> <span class="o">*</span><a
href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">params</span></a><span class="p">)</span><s [...]
+<span class="nb">print</span><span class="p">(</span><a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"forward"</span><span
class="p">](</span><span class="n">tvm_data</span><span class="p">,</span>
<span class="o">*</span><a href="http [...]
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[26480.525 26398.688 24851.756 26216.059
25056.197 26085.629 25934.707
- 25167.797 26442.43 25794.87 ]]
+<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[23709.96 23185.262 22688.762 24297.123
22736.96 23683.799 23711.588
+ 24120.336 23535.51 24328.434]]
</pre></div>
</div>
<p>Our goal is to bring machine learning to the application with any language
of interest,
@@ -468,8 +468,8 @@ with the minimum runtime support.</p>
<ul>
<li><p>Each function in IRModule becomes a runnable function in the runtime.
For example in LLM
cases, we can call <code class="docutils literal notranslate"><span
class="pre">prefill</span></code> and <code class="docutils literal
notranslate"><span class="pre">decode</span></code> functions directly.</p>
-<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><span class="n">prefill_logits</span> <span
class="o">=</span> <span class="n">vm</span><span class="p">[</span><span
class="s2">"prefill"</span><span class="p">](</span><span
class="n">inputs</span><span class="p">,</span> <span
class="n">weight</span><span class="p">,</span> <span
class="n">kv_cache</span><span class="p">)</span>
-<span class="n">decoded_logits</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"decode"</span><span class="p">](</span><span
class="n">inputs</span><span class="p">,</span> <span
class="n">weight</span><span class="p">,</span> <span
class="n">kv_cache</span><span class="p">)</span>
+<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><span class="n">prefill_logits</span> <span
class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"prefill"</span><span
class="p">](</span> [...]
+<span class="n">decoded_logits</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"decode"</span><span
class="p">](</span><span class="n">inputs</span><span class="p">,</span> <span
class="n">weight</span>< [...]
</pre></div>
</div>
</li>
@@ -484,15 +484,15 @@ copy exchange with existing ecosystem (DLPack exchange
with PyTorch)</p>
</li>
<li><p>TVM runtime works in non-python environments, so it works on settings
such as mobile</p>
<div class="highlight-C++ notranslate"><div
class="highlight"><pre><span></span><span class="c1">// C++ snippet</span>
-<span class="n">runtime</span><span class="o">::</span><span
class="n">Module</span><span class="w"> </span><span class="n">vm</span><span
class="w"> </span><span class="o">=</span><span class="w"> </span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">ex</span></a><span class="p">.</span><span class="n">GetFunction [...]
-<span class="n">vm</span><span class="p">.</span><span
class="n">GetFunction</span><span class="p">(</span><span
class="s">"init"</span><span class="p">)(...);</span>
-<span class="n">Tensor</span><span class="w"> </span><span
class="n">out</span><span class="w"> </span><span class="o">=</span><span
class="w"> </span><span class="n">vm</span><span class="p">.</span><span
class="n">GetFunction</span><span class="p">(</span><span
class="s">"prefill"</span><span class="p">)(</span><span
class="n">data</span><span class="p">,</span><span class="w"> </span><span
class="n">weight</span><span class="p">,</span><span class="w"> </span><span
class="n" [...]
+<span class="n">runtime</span><span class="o">::</span><span
class="n">Module</span><span class="w"> </span><a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span class="w">
</span><span class="o">=</span><span class="w"> </span><a
href="../../reference/api/python/relax/relax.html#tvm.r [...]
+<a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">.</span><span class="n">GetFunction</span><span
class="p">(</span><span class="s">"init"</span><span
class="p">)(...);</span>
+<span class="n">Tensor</span><span class="w"> </span><span
class="n">out</span><span class="w"> </span><span class="o">=</span><span
class="w"> </span><a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">.</span><span class="n">GetFunction</span><span
class="p">(</span><span [...]
</pre></div>
</div>
<div class="highlight-Java notranslate"><div
class="highlight"><pre><span></span><span class="c1">// Java snippet</span>
-<span class="n">Module</span><span class="w"> </span><span
class="n">vm</span><span class="w"> </span><span class="o">=</span><span
class="w"> </span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">ex</span></a><span class="p">.</span><span
class="na">getFunction</span><span class="p">(</span><span class="s">"l
[...]
-<span class="n">vm</span><span class="p">.</span><span
class="na">getFunction</span><span class="p">(</span><span
class="s">"init"</span><span class="p">).</span><span
class="na">pushArg</span><span class="p">(...).</span><span
class="na">invoke</span><span class="p">;</span>
-<span class="n">Tensor</span><span class="w"> </span><span
class="n">out</span><span class="w"> </span><span class="o">=</span><span
class="w"> </span><span class="n">vm</span><span class="p">.</span><span
class="na">getFunction</span><span class="p">(</span><span
class="s">"prefill"</span><span class="p">).</span><span
class="na">pushArg</span><span class="p">(</span><span
class="n">data</span><span class="p">).</span><span
class="na">pushArg</span><span class="p">(</span><spa [...]
+<span class="n">Module</span><span class="w"> </span><a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span class="w">
</span><span class="o">=</span><span class="w"> </span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class [...]
+<a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">.</span><span class="na">getFunction</span><span
class="p">(</span><span class="s">"init"</span><span
class="p">).</span><span class="na">pushArg</span><span
class="p">(...).</span><span class="na">invoke</span>< [...]
+<span class="n">Tensor</span><span class="w"> </span><span
class="n">out</span><span class="w"> </span><span class="o">=</span><span
class="w"> </span><a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">.</span><span class="na">getFunction</span><span
class="p">(</span><spa [...]
</pre></div>
</div>
</li>
diff --git a/docs/get_started/tutorials/sg_execution_times.html
b/docs/get_started/tutorials/sg_execution_times.html
index d1d40d9e1d..bd84399a92 100644
--- a/docs/get_started/tutorials/sg_execution_times.html
+++ b/docs/get_started/tutorials/sg_execution_times.html
@@ -293,7 +293,7 @@
<section id="computation-times">
<span
id="sphx-glr-get-started-tutorials-sg-execution-times"></span><h1>Computation
times<a class="headerlink" href="#computation-times" title="Link to this
heading"></a></h1>
-<p><strong>00:10.678</strong> total execution time for 2 files <strong>from
get_started/tutorials</strong>:</p>
+<p><strong>00:10.212</strong> total execution time for 2 files <strong>from
get_started/tutorials</strong>:</p>
<div class="docutils container">
<style scoped>
<link
href="https://cdnjs.cloudflare.com/ajax/libs/twitter-bootstrap/5.3.0/css/bootstrap.min.css"
rel="stylesheet" />
@@ -315,11 +315,11 @@ $(document).ready( function () {
</thead>
<tbody>
<tr class="row-even"><td><p><a class="reference internal"
href="ir_module.html#sphx-glr-get-started-tutorials-ir-module-py"><span
class="std std-ref">IRModule</span></a> (<code class="docutils literal
notranslate"><span class="pre">ir_module.py</span></code>)</p></td>
-<td><p>00:10.477</p></td>
+<td><p>00:10.028</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="quick_start.html#sphx-glr-get-started-tutorials-quick-start-py"><span
class="std std-ref">Quick Start</span></a> (<code class="docutils literal
notranslate"><span class="pre">quick_start.py</span></code>)</p></td>
-<td><p>00:00.201</p></td>
+<td><p>00:00.184</p></td>
<td><p>0.0</p></td>
</tr>
</tbody>
diff --git a/docs/how_to/tutorials/cross_compilation_and_rpc.html
b/docs/how_to/tutorials/cross_compilation_and_rpc.html
index 24d803eb5c..b5808b27c1 100644
--- a/docs/how_to/tutorials/cross_compilation_and_rpc.html
+++ b/docs/how_to/tutorials/cross_compilation_and_rpc.html
@@ -474,7 +474,7 @@ device and returns the measured cost. Network overhead is
excluded.</p>
<span class="nb">print</span><span class="p">(</span><span
class="s2">"</span><span class="si">%g</span><span class="s2">
secs/op"</span> <span class="o">%</span> <span class="n">cost</span><span
class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>1.319e-07 secs/op
+<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>1.2781e-06 secs/op
</pre></div>
</div>
</section>
diff --git a/docs/how_to/tutorials/customize_opt.html
b/docs/how_to/tutorials/customize_opt.html
index 2e4b71c53b..8bfab164ad 100644
--- a/docs/how_to/tutorials/customize_opt.html
+++ b/docs/how_to/tutorials/customize_opt.html
@@ -600,16 +600,16 @@ pushing the performance to the limit. The current
optimization may not be the be
<p>We can build and deploy the optimized model to the TVM runtime.</p>
<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">ex</span></a> <span class="o">=</span> <a
href="../../reference/api/python/driver.html#tvm.compile" title="tvm.compile"
class="sphx-glr-backref-module-tvm sphx-glr-backref-type-py-functi [...]
<span class="n">dev</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">device</span><span
class="p">(</span><span class="s2">"cuda"</span><span
class="p">,</span> <span class="mi">0</span><span class="p">)</span>
-<span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/relax/relax.html#tvm.relax.VirtualMachine"
title="tvm.relax.VirtualMachine" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">relax</span><span class="o">.</span><span
class="n">VirtualMachine</span></a><span class="p">(</span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class [...]
+<a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a> <span
class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm
sphx-glr-backref-type-py-class"><span class=" [...]
<span class="c1"># Need to allocate data and params on GPU device</span>
<span class="n">data</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">runtime</span><span
class="o">.</span><span class="n">tensor</span><span class="p">(</span><span
class="n">np</span><span class="o">.</span><span class="n">random</span><span
class="o">.</span><span class="n">rand</span><span class="p">(</span><span
class="o">*</span><a
href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-ba [...]
<a href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">gpu_params</span></a> <span class="o">=</span> <span
class="p">[</span><span class="n">tvm</span><span class="o">.</span><span
class="n">runtime</span><span class="o">.</span><span
class="n">tensor</span><span class="p">(</span><span class="n">np</span><span
class="o">.</span><span class="n"> [...]
-<span class="n">gpu_out</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"forward"</span><span class="p">](</span><span
class="n">data</span><span class="p">,</span> <span class="o">*</span><a
href="https://docs.python.org/3/library/stdtypes.html#list"
title="builtins.list" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">gpu_params</span></a><span class="p">)</span [...]
+<span class="n">gpu_out</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"forward"</span><span
class="p">](</span><span class="n">data</span><span class="p">,</span> <span
class="o">*</span><a href="https [...]
<span class="nb">print</span><span class="p">(</span><span
class="n">gpu_out</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[23666.377 26352.043 24360.344 25105.71
25748.316 24053.691 25034.05
- 26084.346 24839.422 26120.78 ]]
+<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>[[25804.072 25234.797 24747.855 23863.885
24310.018 26154.75 25283.268
+ 24277.625 25507.771 25346.963]]
</pre></div>
</div>
</section>
diff --git a/docs/how_to/tutorials/e2e_opt_model.html
b/docs/how_to/tutorials/e2e_opt_model.html
index 1f64d972cb..29623477fc 100644
--- a/docs/how_to/tutorials/e2e_opt_model.html
+++ b/docs/how_to/tutorials/e2e_opt_model.html
@@ -330,9 +330,9 @@ PyTorch.</p>
<div class="sphx-glr-script-out highlight-none notranslate"><div
class="highlight"><pre><span></span>Downloading:
"https://download.pytorch.org/models/resnet18-f37072fd.pth" to
/workspace/.cache/torch/hub/checkpoints/resnet18-f37072fd.pth
0%| | 0.00/44.7M [00:00<?, ?B/s]
- 40%|████ | 18.0M/44.7M [00:00<00:00, 188MB/s]
- 92%|█████████▏| 41.0M/44.7M [00:00<00:00, 219MB/s]
-100%|██████████| 44.7M/44.7M [00:00<00:00, 216MB/s]
+ 36%|███▌ | 16.1M/44.7M [00:00<00:00, 169MB/s]
+ 86%|████████▌ | 38.4M/44.7M [00:00<00:00, 207MB/s]
+100%|██████████| 44.7M/44.7M [00:00<00:00, 205MB/s]
</pre></div>
</div>
</section>
@@ -406,7 +406,7 @@ We skip this step in the CI environment.</p>
<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><span class="k">if</span> <span
class="ow">not</span> <a
href="https://docs.python.org/3/library/functions.html#bool"
title="builtins.bool" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">IS_IN_CI</span></a><span class="p">:</span>
<span class="n">ex</span> <span class="o">=</span> <a
href="../../reference/api/python/driver.html#tvm.compile" title="tvm.compile"
class="sphx-glr-backref-module-tvm sphx-glr-backref-type-py-function"><span
class="n">tvm</span><span class="o">.</span><span
class="n">compile</span></a><span class="p">(</span><span
class="n">mod</span><span class="p">,</span> <a
href="../../reference/api/python/target.html#tvm.target.Target"
title="tvm.target.Target" class="sphx-glr-backref-module-tvm [...]
<span class="n">dev</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">device</span><span
class="p">(</span><span class="s2">"cuda"</span><span
class="p">,</span> <span class="mi">0</span><span class="p">)</span>
- <span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/relax/relax.html#tvm.relax.VirtualMachine"
title="tvm.relax.VirtualMachine" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">relax</span><span class="o">.</span><span
class="n">VirtualMachine</span></a><span class="p">(</span><span
class="n">ex</span><span class="p">,</span> <span class="n">dev</span><span
class="p">)</span>
+ <span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm
sphx-glr-backref-type-py-class"><span class="n">relax</span><span
class="o">.</span><span class="n">VirtualMachine</span></a><span
class="p">(</span><span class="n">ex</span><span class="p">,</span> <span
class="n">dev</span><span class="p">)</span>
<span class="c1"># Need to allocate data and params on GPU device</span>
<span class="n">gpu_data</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">runtime</span><span
class="o">.</span><span class="n">tensor</span><span class="p">(</span><span
class="n">np</span><span class="o">.</span><span class="n">random</span><span
class="o">.</span><span class="n">rand</span><span class="p">(</span><span
class="mi">1</span><span class="p">,</span> <span class="mi">3</span><span
class="p">,</span> <span class="mi">224< [...]
<span class="n">gpu_params</span> <span class="o">=</span> <span
class="p">[</span><span class="n">tvm</span><span class="o">.</span><span
class="n">runtime</span><span class="o">.</span><span
class="n">tensor</span><span class="p">(</span><span class="n">p</span><span
class="p">,</span> <span class="n">dev</span><span class="p">)</span> <span
class="k">for</span> <span class="n">p</span> <span class="ow">in</span> <span
class="n">params</span><span class="p">[</span><span class="s2" [...]
diff --git a/docs/how_to/tutorials/optimize_llm.html
b/docs/how_to/tutorials/optimize_llm.html
index 0c46224fed..c66e83000b 100644
--- a/docs/how_to/tutorials/optimize_llm.html
+++ b/docs/how_to/tutorials/optimize_llm.html
@@ -727,7 +727,7 @@ is designed specifically for the LLMs.</p>
<span class="k">with</span> <a
href="../../reference/api/python/target.html#tvm.target.Target"
title="tvm.target.Target" class="sphx-glr-backref-module-tvm-target
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">target</span></a><span class="p">:</span>
<a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">ex</span></a> <span class="o">=</span> <a
href="../../reference/api/python/driver.html#tvm.compile" title="tvm.compile"
class="sphx-glr-backref-module-tvm sphx-glr-backref-type-py-function"><span
class="n">tvm</span><span class="o">.</span><span class="n">compile</ [...]
- <span class="n">vm</span> <span class="o">=</span> <a
href="../../reference/api/python/relax/relax.html#tvm.relax.VirtualMachine"
title="tvm.relax.VirtualMachine" class="sphx-glr-backref-module-tvm-relax
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">relax</span><span class="o">.</span><span
class="n">VirtualMachine</span></a><span class="p">(</span><a
href="../../reference/api/python/relax/relax.html#tvm.relax.VMExecutable"
title="tvm.relax.VMExecutable" c [...]
+ <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a> <span
class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm
sphx-glr-backref-type-py-class"><span cla [...]
</pre></div>
</div>
</section>
@@ -825,7 +825,7 @@ the model documentation for the correct tokenization and
prompt format.</p>
key and value tensors for the attention layer. Apache TVM provides a
PagedKVCache to store the
key and value tensors. We create the PagedKVCache with the specified
parameters.</p>
<div class="highlight-Python notranslate"><div
class="highlight"><pre><span></span><span class="k">if</span> <span
class="ow">not</span> <a
href="https://docs.python.org/3/library/functions.html#bool"
title="builtins.bool" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span
class="n">IS_IN_CI</span></a><span class="p">:</span>
- <span class="n">kv_cache</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"create_tir_paged_kv_cache"</span><span class="p">](</span>
+ <span class="n">kv_cache</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span
class="s2">"create_tir_paged_kv_cache"</span><span class="p">](</span>
<a href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class"><span class="n">ShapeTuple</span></a><span
class="p">([</span><span class="mi">1</span><span class="p">]),</span> <span
class="c1"># max_batch_size=1</span>
<a href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class"><span class="n">ShapeTuple</span></a><span
class="p">([</span><span class="mi">2048</span><span class="p">]),</span>
<span class="c1"># max_total_seq_len=2048</span>
<a href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class"><span class="n">ShapeTuple</span></a><span
class="p">([</span><span class="mi">2048</span><span class="p">]),</span>
<span class="c1"># prefill_chunk_size=2048</span>
@@ -842,7 +842,7 @@ compiled in the Relax IRModule to embed the tokens into the
hidden states.</p>
<span class="k">def</span><span class="w"> </span><span
class="nf">embed</span><span class="p">(</span><span
class="n">tokens</span><span class="p">,</span> <span
class="n">params</span><span class="p">):</span>
- <span class="n">_embed</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"embed"</span><span class="p">](</span><span
class="n">tokens</span><span class="p">,</span> <span
class="n">params</span><span class="p">)</span>
+ <span class="n">_embed</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"embed"</span><span
class="p">](</span><span class="n">tokens</span><span class="p">,</span> <span
class="n">params</span><span [...]
<span class="c1"># Reshape hidden from [seq_len, hidden_size] to [1,
seq_len, hidden_size]</span>
<span class="n">_embed</span> <span class="o">=</span> <span
class="n">nd_view_func</span><span class="p">(</span><span
class="n">_embed</span><span class="p">,</span> <a
href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class"><span class="n">ShapeTuple</span></a><span
class="p">([</span><span class="mi">1</span><span class="p">,</span> <span
class="n">_embed</span><span class="o">.</s [...]
<span class="k">return</span> <span class="n">_embed</span>
@@ -865,7 +865,7 @@ and <cite>end_forward_func</cite> to end the forward
pass.</p>
<span class="n">add_sequence_func</span><span class="p">(</span><span
class="n">kv_cache</span><span class="p">,</span> <span
class="n">seq_id</span><span class="p">)</span>
<span class="n">hidden_states</span> <span class="o">=</span> <span
class="n">embed</span><span class="p">(</span><span
class="n">tokens</span><span class="p">,</span> <span
class="n">params</span><span class="p">)</span>
<span class="n">begin_forward_func</span><span class="p">(</span><span
class="n">kv_cache</span><span class="p">,</span> <a
href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class"><span class="n">ShapeTuple</span></a><span
class="p">([</span><span class="n">seq_id</span><span class="p">]),</span> <a
href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" cla [...]
- <span class="n">logits</span><span class="p">,</span> <span
class="n">kv_cache</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"prefill"</span><span class="p">](</span><span
class="n">hidden_states</span><span class="p">,</span> <span
class="n">kv_cache</span><span class="p">,</span> <span
class="n">params</span><span class="p">)</span>
+ <span class="n">logits</span><span class="p">,</span> <span
class="n">kv_cache</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"prefill"</span><span
class="p">](</span><span class="n">hidden_states</ [...]
<span class="n">end_forward_func</span><span class="p">(</span><span
class="n">kv_cache</span><span class="p">)</span>
</pre></div>
</div>
@@ -897,7 +897,7 @@ IRModule to generate the token.</p>
<span class="n">tokens</span> <span class="o">=</span> <span
class="n">tvm</span><span class="o">.</span><span class="n">runtime</span><span
class="o">.</span><span class="n">tensor</span><span class="p">(</span><span
class="n">np</span><span class="o">.</span><span class="n">array</span><span
class="p">([</span><span class="n">last_token</span><span
class="p">])</span><span class="o">.</span><span class="n">astype</span><span
class="p">(</span><span class="s2">"int32"< [...]
<span class="n">hidden_states</span> <span class="o">=</span> <span
class="n">embed</span><span class="p">(</span><span
class="n">tokens</span><span class="p">,</span> <span
class="n">params</span><span class="p">)</span>
<span class="n">begin_forward_func</span><span class="p">(</span><span
class="n">kv_cache</span><span class="p">,</span> <a
href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" class="sphx-glr-backref-module-builtins
sphx-glr-backref-type-py-class"><span class="n">ShapeTuple</span></a><span
class="p">([</span><span class="n">seq_id</span><span class="p">]),</span> <a
href="https://docs.python.org/3/library/stdtypes.html#tuple"
title="builtins.tuple" [...]
- <span class="n">logits</span><span class="p">,</span> <span
class="n">kv_cache</span> <span class="o">=</span> <span
class="n">vm</span><span class="p">[</span><span
class="s2">"decode"</span><span class="p">](</span><span
class="n">hidden_states</span><span class="p">,</span> <span
class="n">kv_cache</span><span class="p">,</span> <span
class="n">params</span><span class="p">)</span>
+ <span class="n">logits</span><span class="p">,</span> <span
class="n">kv_cache</span> <span class="o">=</span> <a
href="../../reference/api/python/runtime/vm.html#tvm.runtime.vm.VirtualMachine"
title="tvm.runtime.vm.VirtualMachine"
class="sphx-glr-backref-module-tvm-runtime-vm sphx-glr-backref-type-py-class
sphx-glr-backref-instance"><span class="n">vm</span></a><span
class="p">[</span><span class="s2">"decode"</span><span
class="p">](</span><span class="n">hidden_state [...]
<span class="n">end_forward_func</span><span class="p">(</span><span
class="n">kv_cache</span><span class="p">)</span>
<span class="n">last_token</span> <span class="o">=</span> <span
class="n">sample_token</span><span class="p">(</span><span
class="n">logits</span><span class="p">)</span>
diff --git a/docs/how_to/tutorials/sg_execution_times.html
b/docs/how_to/tutorials/sg_execution_times.html
index fc81eba8fa..029893a3f3 100644
--- a/docs/how_to/tutorials/sg_execution_times.html
+++ b/docs/how_to/tutorials/sg_execution_times.html
@@ -293,7 +293,7 @@
<section id="computation-times">
<span id="sphx-glr-how-to-tutorials-sg-execution-times"></span><h1>Computation
times<a class="headerlink" href="#computation-times" title="Link to this
heading"></a></h1>
-<p><strong>00:40.736</strong> total execution time for 4 files <strong>from
how_to/tutorials</strong>:</p>
+<p><strong>00:39.081</strong> total execution time for 4 files <strong>from
how_to/tutorials</strong>:</p>
<div class="docutils container">
<style scoped>
<link
href="https://cdnjs.cloudflare.com/ajax/libs/twitter-bootstrap/5.3.0/css/bootstrap.min.css"
rel="stylesheet" />
@@ -315,19 +315,19 @@ $(document).ready( function () {
</thead>
<tbody>
<tr class="row-even"><td><p><a class="reference internal"
href="optimize_llm.html#sphx-glr-how-to-tutorials-optimize-llm-py"><span
class="std std-ref">Optimize Large Language Model</span></a> (<code
class="docutils literal notranslate"><span
class="pre">optimize_llm.py</span></code>)</p></td>
-<td><p>00:39.050</p></td>
+<td><p>00:37.439</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="e2e_opt_model.html#sphx-glr-how-to-tutorials-e2e-opt-model-py"><span
class="std std-ref">End-to-End Optimize Model</span></a> (<code class="docutils
literal notranslate"><span class="pre">e2e_opt_model.py</span></code>)</p></td>
-<td><p>00:00.768</p></td>
+<td><p>00:00.739</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal"
href="customize_opt.html#sphx-glr-how-to-tutorials-customize-opt-py"><span
class="std std-ref">Customize Optimization</span></a> (<code class="docutils
literal notranslate"><span class="pre">customize_opt.py</span></code>)</p></td>
-<td><p>00:00.708</p></td>
+<td><p>00:00.666</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="cross_compilation_and_rpc.html#sphx-glr-how-to-tutorials-cross-compilation-and-rpc-py"><span
class="std std-ref">Cross Compilation and RPC</span></a> (<code
class="docutils literal notranslate"><span
class="pre">cross_compilation_and_rpc.py</span></code>)</p></td>
-<td><p>00:00.210</p></td>
+<td><p>00:00.237</p></td>
<td><p>0.0</p></td>
</tr>
</tbody>
diff --git a/docs/objects.inv b/docs/objects.inv
index ef6d3196cd..4868121338 100644
Binary files a/docs/objects.inv and b/docs/objects.inv differ
diff --git a/docs/reference/api/python/runtime/vm.html
b/docs/reference/api/python/runtime/vm.html
index 64873232e1..8efe68e506 100644
--- a/docs/reference/api/python/runtime/vm.html
+++ b/docs/reference/api/python/runtime/vm.html
@@ -489,7 +489,7 @@ more details.</p>
<div class="admonition seealso">
<p class="admonition-title">See also</p>
<dl class="simple">
-<dt><a class="reference internal"
href="../relax/relax.html#tvm.relax.VMInstrumentReturnKind"
title="tvm.runtime.vm.VMInstrumentReturnKind"><code class="xref py py-obj
docutils literal notranslate"><span
class="pre">VMInstrumentReturnKind</span></code></a></dt><dd><p>the possible
return values in VM.</p>
+<dt><a class="reference internal"
href="#tvm.runtime.vm.VMInstrumentReturnKind"
title="tvm.runtime.vm.VMInstrumentReturnKind"><code class="xref py py-obj
docutils literal notranslate"><span
class="pre">VMInstrumentReturnKind</span></code></a></dt><dd><p>the possible
return values in VM.</p>
</dd>
</dl>
</div>
diff --git a/docs/searchindex.js b/docs/searchindex.js
index fba0b543ce..7c244ffcd7 100644
--- a/docs/searchindex.js
+++ b/docs/searchindex.js
@@ -1 +1 @@
-Search.setIndex({"alltitles": {"1. Cross Compile TVM Runtime": [[40,
"cross-compile-tvm-runtime"]], "1. The lack of numpy on device machine caused
the RPC server can\u2019t be launched.": [[40,
"the-lack-of-numpy-on-device-machine-caused-the-rpc-server-can-t-be-launched"]],
"2. Pack and Deploy to Device Machine": [[40,
"pack-and-deploy-to-device-machine"]], "2. The lack of cloudpickle on device
machine caused the RPC server can\u2019t be launched.": [[40,
"the-lack-of-cloudpickle-on-devi [...]
\ No newline at end of file
+Search.setIndex({"alltitles": {"1. Cross Compile TVM Runtime": [[40,
"cross-compile-tvm-runtime"]], "1. The lack of numpy on device machine caused
the RPC server can\u2019t be launched.": [[40,
"the-lack-of-numpy-on-device-machine-caused-the-rpc-server-can-t-be-launched"]],
"2. Pack and Deploy to Device Machine": [[40,
"pack-and-deploy-to-device-machine"]], "2. The lack of cloudpickle on device
machine caused the RPC server can\u2019t be launched.": [[40,
"the-lack-of-cloudpickle-on-devi [...]
\ No newline at end of file
diff --git a/docs/sg_execution_times.html b/docs/sg_execution_times.html
index 1be37ba3fa..7cef8b9aef 100644
--- a/docs/sg_execution_times.html
+++ b/docs/sg_execution_times.html
@@ -293,7 +293,7 @@
<section id="computation-times">
<span id="sphx-glr-sg-execution-times"></span><h1>Computation times<a
class="headerlink" href="#computation-times" title="Link to this
heading"></a></h1>
-<p><strong>00:52.122</strong> total execution time for 10 files <strong>from
all galleries</strong>:</p>
+<p><strong>00:49.994</strong> total execution time for 10 files <strong>from
all galleries</strong>:</p>
<div class="docutils container">
<style scoped>
<link
href="https://cdnjs.cloudflare.com/ajax/libs/twitter-bootstrap/5.3.0/css/bootstrap.min.css"
rel="stylesheet" />
@@ -315,43 +315,43 @@ $(document).ready( function () {
</thead>
<tbody>
<tr class="row-even"><td><p><a class="reference internal"
href="how_to/tutorials/optimize_llm.html#sphx-glr-how-to-tutorials-optimize-llm-py"><span
class="std std-ref">Optimize Large Language Model</span></a> (<code
class="docutils literal notranslate"><span
class="pre">../how_to/tutorials/optimize_llm.py</span></code>)</p></td>
-<td><p>00:39.050</p></td>
+<td><p>00:37.439</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="get_started/tutorials/ir_module.html#sphx-glr-get-started-tutorials-ir-module-py"><span
class="std std-ref">IRModule</span></a> (<code class="docutils literal
notranslate"><span
class="pre">../get_started/tutorials/ir_module.py</span></code>)</p></td>
-<td><p>00:10.477</p></td>
+<td><p>00:10.028</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal"
href="how_to/tutorials/e2e_opt_model.html#sphx-glr-how-to-tutorials-e2e-opt-model-py"><span
class="std std-ref">End-to-End Optimize Model</span></a> (<code
class="docutils literal notranslate"><span
class="pre">../how_to/tutorials/e2e_opt_model.py</span></code>)</p></td>
-<td><p>00:00.768</p></td>
+<td><p>00:00.739</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="how_to/tutorials/customize_opt.html#sphx-glr-how-to-tutorials-customize-opt-py"><span
class="std std-ref">Customize Optimization</span></a> (<code class="docutils
literal notranslate"><span
class="pre">../how_to/tutorials/customize_opt.py</span></code>)</p></td>
-<td><p>00:00.708</p></td>
+<td><p>00:00.666</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal"
href="deep_dive/tensor_ir/tutorials/tir_transformation.html#sphx-glr-deep-dive-tensor-ir-tutorials-tir-transformation-py"><span
class="std std-ref">Transformation</span></a> (<code class="docutils literal
notranslate"><span
class="pre">../deep_dive/tensor_ir/tutorials/tir_transformation.py</span></code>)</p></td>
-<td><p>00:00.312</p></td>
+<td><p>00:00.309</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="how_to/tutorials/cross_compilation_and_rpc.html#sphx-glr-how-to-tutorials-cross-compilation-and-rpc-py"><span
class="std std-ref">Cross Compilation and RPC</span></a> (<code
class="docutils literal notranslate"><span
class="pre">../how_to/tutorials/cross_compilation_and_rpc.py</span></code>)</p></td>
-<td><p>00:00.210</p></td>
+<td><p>00:00.237</p></td>
<td><p>0.0</p></td>
</tr>
-<tr class="row-even"><td><p><a class="reference internal"
href="get_started/tutorials/quick_start.html#sphx-glr-get-started-tutorials-quick-start-py"><span
class="std std-ref">Quick Start</span></a> (<code class="docutils literal
notranslate"><span
class="pre">../get_started/tutorials/quick_start.py</span></code>)</p></td>
-<td><p>00:00.201</p></td>
+<tr class="row-even"><td><p><a class="reference internal"
href="deep_dive/tensor_ir/tutorials/tir_creation.html#sphx-glr-deep-dive-tensor-ir-tutorials-tir-creation-py"><span
class="std std-ref">TensorIR Creation</span></a> (<code class="docutils
literal notranslate"><span
class="pre">../deep_dive/tensor_ir/tutorials/tir_creation.py</span></code>)</p></td>
+<td><p>00:00.188</p></td>
<td><p>0.0</p></td>
</tr>
-<tr class="row-odd"><td><p><a class="reference internal"
href="deep_dive/tensor_ir/tutorials/tir_creation.html#sphx-glr-deep-dive-tensor-ir-tutorials-tir-creation-py"><span
class="std std-ref">TensorIR Creation</span></a> (<code class="docutils
literal notranslate"><span
class="pre">../deep_dive/tensor_ir/tutorials/tir_creation.py</span></code>)</p></td>
-<td><p>00:00.189</p></td>
+<tr class="row-odd"><td><p><a class="reference internal"
href="get_started/tutorials/quick_start.html#sphx-glr-get-started-tutorials-quick-start-py"><span
class="std std-ref">Quick Start</span></a> (<code class="docutils literal
notranslate"><span
class="pre">../get_started/tutorials/quick_start.py</span></code>)</p></td>
+<td><p>00:00.184</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal"
href="deep_dive/relax/tutorials/relax_creation.html#sphx-glr-deep-dive-relax-tutorials-relax-creation-py"><span
class="std std-ref">Relax Creation</span></a> (<code class="docutils literal
notranslate"><span
class="pre">../deep_dive/relax/tutorials/relax_creation.py</span></code>)</p></td>
-<td><p>00:00.131</p></td>
+<td><p>00:00.127</p></td>
<td><p>0.0</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal"
href="deep_dive/relax/tutorials/relax_transformation.html#sphx-glr-deep-dive-relax-tutorials-relax-transformation-py"><span
class="std std-ref">Transformation</span></a> (<code class="docutils literal
notranslate"><span
class="pre">../deep_dive/relax/tutorials/relax_transformation.py</span></code>)</p></td>
-<td><p>00:00.075</p></td>
+<td><p>00:00.077</p></td>
<td><p>0.0</p></td>
</tr>
</tbody>