You are viewing a plain text version of this content. The canonical link for it is here.
Posted to commits@tvm.apache.org by tq...@apache.org on 2022/12/13 17:12:46 UTC
[tvm-site] branch asf-site updated: deploying docs (apache/tvm@12311dcdefd7f2213ce5ce78f2c590444a04b32d)
This is an automated email from the ASF dual-hosted git repository.
tqchen pushed a commit to branch asf-site
in repository https://gitbox.apache.org/repos/asf/tvm-site.git
The following commit(s) were added to refs/heads/asf-site by this push:
new 7deb6f8d51 deploying docs (apache/tvm@12311dcdefd7f2213ce5ce78f2c590444a04b32d)
7deb6f8d51 is described below
commit 7deb6f8d51634aa1355a800e36f01fbf6869e3cb
Author: tvm-bot <95...@users.noreply.github.com>
AuthorDate: Tue Dec 13 17:12:38 2022 +0000
deploying docs (apache/tvm@12311dcdefd7f2213ce5ce78f2c590444a04b32d)
---
docs/_images/sphx_glr_micro_train_001.png | Bin 322550 -> 332113 bytes
docs/_images/sphx_glr_micro_train_thumb.png | Bin 23209 -> 23380 bytes
.../how_to/compile_models/from_darknet.rst.txt | 2 +-
.../how_to/compile_models/from_keras.rst.txt | 2 +-
.../how_to/compile_models/from_mxnet.rst.txt | 2 +-
.../how_to/compile_models/from_oneflow.rst.txt | 2 +-
.../how_to/compile_models/from_pytorch.rst.txt | 2 +-
.../how_to/compile_models/from_tensorflow.rst.txt | 2 +-
.../compile_models/sg_execution_times.rst.txt | 22 +-
.../deploy_models/deploy_model_on_adreno.rst.txt | 2 +-
.../deploy_models/deploy_model_on_android.rst.txt | 2 +-
.../deploy_object_detection_pytorch.rst.txt | 4 +-
.../deploy_models/deploy_prequantized.rst.txt | 6 +-
.../deploy_prequantized_tflite.rst.txt | 4 +-
.../how_to/deploy_models/deploy_quantized.rst.txt | 2 +-
.../deploy_models/deploy_ssd_gluoncv.rst.txt | 4 +-
.../deploy_models/sg_execution_times.rst.txt | 20 +-
.../extend_tvm/bring_your_own_datatypes.rst.txt | 2 +-
.../how_to/extend_tvm/sg_execution_times.rst.txt | 8 +-
.../how_to/extend_tvm/use_pass_instrument.rst.txt | 16 +-
.../optimize_operators/opt_conv_cuda.rst.txt | 2 +-
.../optimize_operators/opt_conv_tensorcore.rst.txt | 2 +-
.../how_to/optimize_operators/opt_gemm.rst.txt | 16 +-
.../optimize_operators/sg_execution_times.rst.txt | 8 +-
.../sg_execution_times.rst.txt | 14 +-
.../tune_conv2d_layer_cuda.rst.txt | 2659 ++------------------
.../tune_network_cuda.rst.txt | 4 +-
.../tune_network_x86.rst.txt | 4 +-
.../tune_sparse_x86.rst.txt | 346 +--
.../tune_with_autotvm/sg_execution_times.rst.txt | 10 +-
.../tune_with_autotvm/tune_conv2d_cuda.rst.txt | 316 ++-
.../work_with_microtvm/micro_autotune.rst.txt | 16 +-
.../work_with_microtvm/micro_pytorch.rst.txt | 4 +-
.../how_to/work_with_microtvm/micro_train.rst.txt | 18 +-
.../work_with_microtvm/sg_execution_times.rst.txt | 12 +-
.../work_with_relay/sg_execution_times.rst.txt | 8 +-
.../how_to/work_with_schedules/intrin_math.rst.txt | 2 +-
.../work_with_schedules/sg_execution_times.rst.txt | 16 +-
.../how_to/work_with_schedules/tensorize.rst.txt | 2 +-
.../tutorials/autotvm/sg_execution_times.rst.txt | 4 +-
.../frontend/deploy_classification.rst.txt | 2 +-
.../tutorials/frontend/deploy_detection.rst.txt | 2 +-
.../tutorials/frontend/sg_execution_times.rst.txt | 6 +-
.../tutorials/optimize/sg_execution_times.rst.txt | 6 +-
.../topic/vta/tutorials/sg_execution_times.rst.txt | 6 +-
.../tutorial/auto_scheduler_matmul_x86.rst.txt | 13 +-
docs/_sources/tutorial/autotvm_matmul_x86.rst.txt | 20 +-
docs/_sources/tutorial/autotvm_relay_x86.rst.txt | 58 +-
.../tutorial/cross_compilation_and_rpc.rst.txt | 2 +-
docs/_sources/tutorial/intro_topi.rst.txt | 2 +-
docs/_sources/tutorial/sg_execution_times.rst.txt | 20 +-
.../tutorial/tensor_expr_get_started.rst.txt | 49 +-
docs/commit_hash | 2 +-
docs/how_to/compile_models/from_darknet.html | 2 +-
docs/how_to/compile_models/from_keras.html | 2 +-
docs/how_to/compile_models/from_mxnet.html | 2 +-
docs/how_to/compile_models/from_oneflow.html | 13 +-
docs/how_to/compile_models/from_pytorch.html | 10 +-
docs/how_to/compile_models/from_tensorflow.html | 2 +-
docs/how_to/compile_models/sg_execution_times.html | 26 +-
.../deploy_models/deploy_model_on_adreno.html | 2 +-
.../deploy_models/deploy_model_on_android.html | 2 +-
.../deploy_object_detection_pytorch.html | 46 +-
docs/how_to/deploy_models/deploy_prequantized.html | 9 +-
.../deploy_models/deploy_prequantized_tflite.html | 4 +-
docs/how_to/deploy_models/deploy_quantized.html | 2 +-
docs/how_to/deploy_models/deploy_ssd_gluoncv.html | 37 +-
docs/how_to/deploy_models/sg_execution_times.html | 20 +-
.../extend_tvm/bring_your_own_datatypes.html | 2 +-
docs/how_to/extend_tvm/sg_execution_times.html | 8 +-
docs/how_to/extend_tvm/use_pass_instrument.html | 16 +-
docs/how_to/optimize_operators/opt_conv_cuda.html | 2 +-
.../optimize_operators/opt_conv_tensorcore.html | 2 +-
docs/how_to/optimize_operators/opt_gemm.html | 16 +-
.../optimize_operators/sg_execution_times.html | 8 +-
.../sg_execution_times.html | 14 +-
.../tune_conv2d_layer_cuda.html | 2659 ++------------------
.../tune_with_autoscheduler/tune_network_cuda.html | 4 +-
.../tune_with_autoscheduler/tune_network_x86.html | 4 +-
.../tune_with_autoscheduler/tune_sparse_x86.html | 346 +--
.../tune_with_autotvm/sg_execution_times.html | 10 +-
.../how_to/tune_with_autotvm/tune_conv2d_cuda.html | 316 ++-
docs/how_to/work_with_microtvm/micro_autotune.html | 16 +-
docs/how_to/work_with_microtvm/micro_pytorch.html | 5 +-
docs/how_to/work_with_microtvm/micro_train.html | 16 +-
.../work_with_microtvm/sg_execution_times.html | 12 +-
.../how_to/work_with_relay/sg_execution_times.html | 8 +-
docs/how_to/work_with_schedules/intrin_math.html | 2 +-
.../work_with_schedules/sg_execution_times.html | 16 +-
docs/how_to/work_with_schedules/tensorize.html | 2 +-
docs/reference/api/python/auto_scheduler.html | 4 +-
docs/reference/api/python/topi.html | 2 +-
.../api/typedoc/classes/bytestreamreader.html | 12 +-
.../api/typedoc/classes/cachedcallstack.html | 34 +-
docs/reference/api/typedoc/classes/dldatatype.html | 12 +-
docs/reference/api/typedoc/classes/dldevice.html | 10 +-
.../reference/api/typedoc/classes/environment.html | 12 +-
docs/reference/api/typedoc/classes/ffilibrary.html | 20 +-
.../api/typedoc/classes/graphexecutor.html | 16 +-
docs/reference/api/typedoc/classes/instance.html | 40 +-
docs/reference/api/typedoc/classes/memory.html | 34 +-
docs/reference/api/typedoc/classes/module.html | 10 +-
docs/reference/api/typedoc/classes/ndarray.html | 22 +-
.../api/typedoc/classes/packedfunccell.html | 6 +-
docs/reference/api/typedoc/classes/rpcserver.html | 14 +-
docs/reference/api/typedoc/classes/scalar.html | 6 +-
.../api/typedoc/classes/webgpucontext.html | 12 +-
docs/reference/api/typedoc/enums/argtypecode.html | 30 +-
.../api/typedoc/enums/aynccallbackcode.html | 4 +-
.../api/typedoc/enums/dldatatypecode.html | 8 +-
.../api/typedoc/enums/rpcserverstate.html | 12 +-
docs/reference/api/typedoc/enums/sizeof.html | 18 +-
docs/reference/api/typedoc/index.html | 112 +-
.../api/typedoc/interfaces/disposable.html | 2 +-
.../api/typedoc/interfaces/functioninfo.html | 6 +-
.../api/typedoc/interfaces/libraryprovider.html | 4 +-
docs/searchindex.js | 2 +-
.../vta/tutorials/autotvm/sg_execution_times.html | 4 +-
.../tutorials/frontend/deploy_classification.html | 2 +-
.../vta/tutorials/frontend/deploy_detection.html | 2 +-
.../vta/tutorials/frontend/sg_execution_times.html | 6 +-
.../vta/tutorials/optimize/sg_execution_times.html | 6 +-
docs/topic/vta/tutorials/sg_execution_times.html | 6 +-
docs/tutorial/auto_scheduler_matmul_x86.html | 8 +-
docs/tutorial/autotvm_matmul_x86.html | 20 +-
docs/tutorial/autotvm_relay_x86.html | 270 +-
docs/tutorial/cross_compilation_and_rpc.html | 2 +-
docs/tutorial/intro_topi.html | 2 +-
docs/tutorial/sg_execution_times.html | 20 +-
docs/tutorial/tensor_expr_get_started.html | 45 +-
130 files changed, 1844 insertions(+), 6431 deletions(-)
diff --git a/docs/_images/sphx_glr_micro_train_001.png b/docs/_images/sphx_glr_micro_train_001.png
index b961179557..0fd9c0ef04 100644
Binary files a/docs/_images/sphx_glr_micro_train_001.png and b/docs/_images/sphx_glr_micro_train_001.png differ
diff --git a/docs/_images/sphx_glr_micro_train_thumb.png b/docs/_images/sphx_glr_micro_train_thumb.png
index ab1b531c46..2b4fe4d842 100644
Binary files a/docs/_images/sphx_glr_micro_train_thumb.png and b/docs/_images/sphx_glr_micro_train_thumb.png differ
diff --git a/docs/_sources/how_to/compile_models/from_darknet.rst.txt b/docs/_sources/how_to/compile_models/from_darknet.rst.txt
index 378362410e..150578d16b 100644
--- a/docs/_sources/how_to/compile_models/from_darknet.rst.txt
+++ b/docs/_sources/how_to/compile_models/from_darknet.rst.txt
@@ -315,7 +315,7 @@ The process is no different from other examples.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 9.325 seconds)
+ **Total running time of the script:** ( 1 minutes 10.274 seconds)
.. _sphx_glr_download_how_to_compile_models_from_darknet.py:
diff --git a/docs/_sources/how_to/compile_models/from_keras.rst.txt b/docs/_sources/how_to/compile_models/from_keras.rst.txt
index 77f77d7e0f..d6841bde26 100644
--- a/docs/_sources/how_to/compile_models/from_keras.rst.txt
+++ b/docs/_sources/how_to/compile_models/from_keras.rst.txt
@@ -228,7 +228,7 @@ Look up prediction top 1 index in 1000 class synset.
.. code-block:: none
Relay top-1 id: 285, class name: Egyptian cat
-
1/1 [==============================] - ETA: 0s
1/1 [==============================] - 1s 925ms/step
+
1/1 [==============================] - ETA: 0s
1/1 [==============================] - 1s 956ms/step
Keras top-1 id: 285, class name: Egyptian cat
diff --git a/docs/_sources/how_to/compile_models/from_mxnet.rst.txt b/docs/_sources/how_to/compile_models/from_mxnet.rst.txt
index 420b9b68b1..a16f076ffd 100644
--- a/docs/_sources/how_to/compile_models/from_mxnet.rst.txt
+++ b/docs/_sources/how_to/compile_models/from_mxnet.rst.txt
@@ -115,7 +115,7 @@ In this section, we download a pretrained imagenet model and classify an image.
.. code-block:: none
- Downloading /workspace/.mxnet/models/resnet18_v1-a0666292.zipc507092a-21dc-4fa0-b0af-61eeba7ff005 from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/resnet18_v1-a0666292.zip...
+ Downloading /workspace/.mxnet/models/resnet18_v1-a0666292.zip02d83f1a-3f5a-43d2-b737-d5af5d3a6d59 from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/resnet18_v1-a0666292.zip...
x (1, 3, 224, 224)
diff --git a/docs/_sources/how_to/compile_models/from_oneflow.rst.txt b/docs/_sources/how_to/compile_models/from_oneflow.rst.txt
index fbfb8f1821..5c0f58fc77 100644
--- a/docs/_sources/how_to/compile_models/from_oneflow.rst.txt
+++ b/docs/_sources/how_to/compile_models/from_oneflow.rst.txt
@@ -116,7 +116,7 @@ Load a pretrained OneFlow model and save model
.. code-block:: none
Downloading: "https://oneflow-public.oss-cn-beijing.aliyuncs.com/model_zoo/flowvision/classification/ResNet/resnet18.zip" to /workspace/.oneflow/flowvision_cache/resnet18.zip
-
0%| | 0.00/41.5M [00:00<?, ?B/s]
19%|#9 | 7.99M/41.5M [00:00<00:00, 45.8MB/s]
39%|###8 | 16.0M/41.5M [00:00<00:00, 47.6MB/s]
58%|#####7 | 24.0M/41.5M [00:00<00:00, 49.1MB/s]
77%|#######7 | 32.0M/41.5M [00:00<00:00, 54.5MB/s]
96%|#########6| 40.0M/41.5M [00:00<00:00, 58.5MB/s]
100%|##########| 41.5M/41.5M [00:00<00:00, 55.9MB/s]
+
0%| | 0.00/41.5M [00:00<?, ?B/s]
17%|#6 | 6.95M/41.5M [00:00<00:00, 72.8MB/s]
33%|###3 | 13.9M/41.5M [00:00<00:00, 65.1MB/s]
49%|####8 | 20.2M/41.5M [00:00<00:00, 37.5MB/s]
59%|#####9 | 24.6M/41.5M [00:00<00:00, 36.8MB/s]
77%|#######7 | 32.0M/41.5M [00:00<00:00, 46.1MB/s]
92%|#########2| 38.3M/41.5M [00:00<00:00, 39.8MB/s]
100%|##########| 41.5M/41.5M [00:01<00:00, 43.3MB/s]
diff --git a/docs/_sources/how_to/compile_models/from_pytorch.rst.txt b/docs/_sources/how_to/compile_models/from_pytorch.rst.txt
index 3dc34d36a6..502c3932ab 100644
--- a/docs/_sources/how_to/compile_models/from_pytorch.rst.txt
+++ b/docs/_sources/how_to/compile_models/from_pytorch.rst.txt
@@ -98,7 +98,7 @@ Load a pretrained PyTorch model
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torchvision/models/_utils.py:223: UserWarning: Arguments other than a weight enum or `None` for 'weights' are deprecated since 0.13 and will be removed in 0.15. The current behavior is equivalent to passing `weights=ResNet18_Weights.IMAGENET1K_V1`. You can also use `weights=ResNet18_Weights.DEFAULT` to get the most up-to-date weights.
warnings.warn(msg)
Downloading: "https://download.pytorch.org/models/resnet18-f37072fd.pth" to /workspace/.cache/torch/hub/checkpoints/resnet18-f37072fd.pth
-
0%| | 0.00/44.7M [00:00<?, ?B/s]
18%|#7 | 7.99M/44.7M [00:00<00:00, 60.4MB/s]
54%|#####3 | 24.0M/44.7M [00:00<00:00, 108MB/s]
78%|#######7 | 34.8M/44.7M [00:00<00:00, 103MB/s]
100%|##########| 44.7M/44.7M [00:00<00:00, 94.8MB/s]
+
0%| | 0.00/44.7M [00:00<?, ?B/s]
18%|#7 | 7.99M/44.7M [00:00<00:00, 49.2MB/s]
35%|###5 | 15.7M/44.7M [00:00<00:00, 52.0MB/s]
54%|#####3 | 24.0M/44.7M [00:00<00:00, 58.1MB/s]
72%|#######1 | 32.0M/44.7M [00:00<00:00, 60.4MB/s]
94%|#########4| 42.1M/44.7M [00:00<00:00, 72.8MB/s]
100%|##########| 44.7M/44.7M [00:00<00:00, 65.2MB/s]
diff --git a/docs/_sources/how_to/compile_models/from_tensorflow.rst.txt b/docs/_sources/how_to/compile_models/from_tensorflow.rst.txt
index 2b3d58e594..5bd14cd62b 100644
--- a/docs/_sources/how_to/compile_models/from_tensorflow.rst.txt
+++ b/docs/_sources/how_to/compile_models/from_tensorflow.rst.txt
@@ -416,7 +416,7 @@ Run the corresponding model on tensorflow
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 11.613 seconds)
+ **Total running time of the script:** ( 1 minutes 12.870 seconds)
.. _sphx_glr_download_how_to_compile_models_from_tensorflow.py:
diff --git a/docs/_sources/how_to/compile_models/sg_execution_times.rst.txt b/docs/_sources/how_to/compile_models/sg_execution_times.rst.txt
index 5c23fb0dc8..119196822d 100644
--- a/docs/_sources/how_to/compile_models/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/compile_models/sg_execution_times.rst.txt
@@ -5,26 +5,26 @@
Computation times
=================
-**05:41.464** total execution time for **how_to_compile_models** files:
+**05:46.830** total execution time for **how_to_compile_models** files:
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_tensorflow.py` (``from_tensorflow.py``) | 01:11.613 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_tensorflow.py` (``from_tensorflow.py``) | 01:12.870 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_darknet.py` (``from_darknet.py``) | 01:09.325 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_darknet.py` (``from_darknet.py``) | 01:10.274 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_paddle.py` (``from_paddle.py``) | 00:46.150 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_paddle.py` (``from_paddle.py``) | 00:46.820 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_oneflow.py` (``from_oneflow.py``) | 00:32.015 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_oneflow.py` (``from_oneflow.py``) | 00:33.016 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_mxnet.py` (``from_mxnet.py``) | 00:28.622 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_mxnet.py` (``from_mxnet.py``) | 00:28.659 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_coreml.py` (``from_coreml.py``) | 00:26.501 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_tflite.py` (``from_tflite.py``) | 00:26.100 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_tflite.py` (``from_tflite.py``) | 00:25.450 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_coreml.py` (``from_coreml.py``) | 00:26.060 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_pytorch.py` (``from_pytorch.py``) | 00:22.198 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_pytorch.py` (``from_pytorch.py``) | 00:22.672 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_keras.py` (``from_keras.py``) | 00:17.209 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_keras.py` (``from_keras.py``) | 00:17.903 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_compile_models_from_onnx.py` (``from_onnx.py``) | 00:02.383 | 0.0 MB |
+| :ref:`sphx_glr_how_to_compile_models_from_onnx.py` (``from_onnx.py``) | 00:02.456 | 0.0 MB |
+-----------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/deploy_models/deploy_model_on_adreno.rst.txt b/docs/_sources/how_to/deploy_models/deploy_model_on_adreno.rst.txt
index bd609bf172..22ee141d9e 100644
--- a/docs/_sources/how_to/deploy_models/deploy_model_on_adreno.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_model_on_adreno.rst.txt
@@ -723,7 +723,7 @@ well as provides information about the model's performance
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 2520.0950 2518.8338 2534.8240 2515.2016 5.4340
+ 2518.6538 2517.1939 2526.2019 2515.5964 2.9954
diff --git a/docs/_sources/how_to/deploy_models/deploy_model_on_android.rst.txt b/docs/_sources/how_to/deploy_models/deploy_model_on_android.rst.txt
index 20d44cf378..cfc57e814f 100644
--- a/docs/_sources/how_to/deploy_models/deploy_model_on_android.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_model_on_android.rst.txt
@@ -433,7 +433,7 @@ Execute on TVM
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 15.6967 15.6694 15.9157 15.5146 0.1490
+ 16.0341 16.0132 16.2077 15.9043 0.0924
diff --git a/docs/_sources/how_to/deploy_models/deploy_object_detection_pytorch.rst.txt b/docs/_sources/how_to/deploy_models/deploy_object_detection_pytorch.rst.txt
index d9111a2e01..2fc8cafdaa 100644
--- a/docs/_sources/how_to/deploy_models/deploy_object_detection_pytorch.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_object_detection_pytorch.rst.txt
@@ -127,7 +127,7 @@ Load pre-trained maskrcnn from torchvision and do tracing
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torchvision/models/_utils.py:223: UserWarning: Arguments other than a weight enum or `None` for 'weights' are deprecated since 0.13 and will be removed in 0.15. The current behavior is equivalent to passing `weights=MaskRCNN_ResNet50_FPN_Weights.COCO_V1`. You can also use `weights=MaskRCNN_ResNet50_FPN_Weights.DEFAULT` to get the most up-to-date weights.
warnings.warn(msg)
Downloading: "https://download.pytorch.org/models/maskrcnn_resnet50_fpn_coco-bf2d0c1e.pth" to /workspace/.cache/torch/hub/checkpoints/maskrcnn_resnet50_fpn_coco-bf2d0c1e.pth
-
0%| | 0.00/170M [00:00<?, ?B/s]
5%|4 | 8.12M/170M [00:00<00:02, 83.3MB/s]
11%|#1 | 19.2M/170M [00:00<00:01, 102MB/s]
17%|#7 | 29.0M/170M [00:00<00:01, 98.1MB/s]
23%|##2 | 38.4M/170M [00:00<00:01, 81.2MB/s]
27%|##7 | 46.4M/170M [00:00<00:01, 82.2MB/s]
33%|###2 | 56.0M/170M [00:00<00:01, 72.9MB/s]
38%|###7 | 64.0M/170M [00:00<00:01, 74.8MB/s]
42%|####2 | 72.0M/170M [00:00<00:01, 77.0MB/s]
47%|####7 | 80.0M/170M [00:01<00:01, 77.9MB/s]
52%|#####1 | 88.0M/170M [00:01<00:01, 78.1MB/s]
57%|#####6 | 96.0M/170M [00:01<00:00, 77.8MB/s]
61%|######1 | 104M/170M [00:01<00:00, 70.0MB/s]
66%|######5 | 112M/170M [00:01<00:00, 68.0MB/s]
71%|#######1 | 121M/170M [00:01<00:00, 75.3MB/s]
76%|#######5 | 129M/170M [00:01<00:00, 63.9MB/s]
80%|######## | 136M/170M [00:01<00:00, 65.3MB/s]
85%|########4 | 144M/170M [00:02<00:00, 68.5MB/s]
89%|########9 | 152M/170M [00:02<00:00, 61.3MB/s]
94%|#########4| 160M/170M [00:02<00:00, 60.7MB/s]
99%|#########8| 168M/170M [00:02<00:00, 66.6MB/s]
100%|##########| 170M/170M [00:02<00:00, 72.6MB/s]
+
0%| | 0.00/170M [00:00<?, ?B/s]
5%|4 | 7.99M/170M [00:00<00:02, 75.0MB/s]
9%|8 | 15.2M/170M [00:00<00:02, 73.0MB/s]
13%|#3 | 22.1M/170M [00:00<00:02, 72.3MB/s]
17%|#7 | 29.0M/170M [00:00<00:02, 52.1MB/s]
20%|## | 34.5M/170M [00:01<00:05, 23.7MB/s]
26%|##6 | 44.4M/170M [00:01<00:03, 35.8MB/s]
31%|### | 52.1M/170M [00:01<00:02, 43.8MB/s]
35%|###4 | 58.7M/170M [00:01<00:02, 42.7MB/s]
38%|###7 | 64.3M/170M [00:01<00:02, 42.3MB/s]
44%|####3 | 74.1M/170M [00:01<00:02, 49.7MB/s]
48%|####8 | 82.1M/170M [00:01<00:01, 54.9MB/s]
52%|#####1 | 88.0M/170M [00:02<00:01, 45.8MB/s]
57%|#####6 | 96.1M/170M [00:02<00:01, 51.2MB/s]
61%|######1 | 104M/170M [00:02<00:01, 48.0MB/s]
67%|######7 | 114M/170M [00:02<00:00, 59.5MB/s]
71%|####### | 120M/170M [00:02<00:00, 52.8MB/s]
75%|#######5 | 128M/170M [00:02<00:00, 54.6MB/s]
80%|######## | 136M/170M [00:02<00:00, 52.6MB/s]
86%|########5 | 145M/170M [00:03<00:00, 62.2MB/s]
89%|########9 | 152M/170M [00:03<00:00, 57.2MB/s]
93%|#########2| 158M/170M [00:03<00:00, 57.9MB/s]
96%|#########6| 164M/170M [00:03<00:00, 53.0MB/s]
100%|##########| 170M/170M [00:03<00:00, 50.7MB/s]
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torch/nn/functional.py:3897: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.clone().detach() or sourceTensor.clone().detach().requires_grad_(True), rather than torch.tensor(sourceTensor).
for i in range(dim)
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torchvision/models/detection/anchor_utils.py:124: UserWarning: __floordiv__ is deprecated, and its behavior will change in a future version of pytorch. It currently rounds toward 0 (like the 'trunc' function NOT 'floor'). This results in incorrect rounding for negative values. To keep the current behavior, use torch.div(a, b, rounding_mode='trunc'), or for actual floor division, use torch.div(a, b, rounding_mode='floor').
@@ -296,7 +296,7 @@ Get boxes with score larger than 0.9
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 3 minutes 12.782 seconds)
+ **Total running time of the script:** ( 3 minutes 16.741 seconds)
.. _sphx_glr_download_how_to_deploy_models_deploy_object_detection_pytorch.py:
diff --git a/docs/_sources/how_to/deploy_models/deploy_prequantized.rst.txt b/docs/_sources/how_to/deploy_models/deploy_prequantized.rst.txt
index cf01623e51..131b4c0276 100644
--- a/docs/_sources/how_to/deploy_models/deploy_prequantized.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_prequantized.rst.txt
@@ -236,7 +236,7 @@ training. Other models require a full post training calibration.
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torchvision/models/_utils.py:223: UserWarning: Arguments other than a weight enum or `None` for 'weights' are deprecated since 0.13 and will be removed in 0.15. The current behavior is equivalent to passing `weights=MobileNet_V2_Weights.IMAGENET1K_V1`. You can also use `weights=MobileNet_V2_Weights.DEFAULT` to get the most up-to-date weights.
warnings.warn(msg)
Downloading: "https://download.pytorch.org/models/mobilenet_v2-b0353104.pth" to /workspace/.cache/torch/hub/checkpoints/mobilenet_v2-b0353104.pth
-
0%| | 0.00/13.6M [00:00<?, ?B/s]
59%|#####8 | 7.99M/13.6M [00:00<00:00, 58.8MB/s]
100%|##########| 13.6M/13.6M [00:00<00:00, 75.0MB/s]
+
0%| | 0.00/13.6M [00:00<?, ?B/s]
59%|#####8 | 7.99M/13.6M [00:00<00:00, 48.3MB/s]
93%|#########2| 12.6M/13.6M [00:00<00:00, 40.7MB/s]
100%|##########| 13.6M/13.6M [00:00<00:00, 44.5MB/s]
@@ -418,7 +418,7 @@ Here we give an example of how to measure performance of TVM compiled models.
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 90.5552 90.3865 100.4027 90.0171 1.1241
+ 90.3728 90.2583 93.4944 90.0504 0.4752
@@ -467,7 +467,7 @@ TODO
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 6.105 seconds)
+ **Total running time of the script:** ( 1 minutes 6.981 seconds)
.. _sphx_glr_download_how_to_deploy_models_deploy_prequantized.py:
diff --git a/docs/_sources/how_to/deploy_models/deploy_prequantized_tflite.rst.txt b/docs/_sources/how_to/deploy_models/deploy_prequantized_tflite.rst.txt
index 94496c92c0..c723e6be2e 100644
--- a/docs/_sources/how_to/deploy_models/deploy_prequantized_tflite.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_prequantized_tflite.rst.txt
@@ -432,7 +432,7 @@ Here we give an example of how to measure performance of TVM compiled models.
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 121.0501 120.9877 124.7303 120.0469 0.5382
+ 123.0244 122.9984 125.3565 122.1152 0.5555
@@ -469,7 +469,7 @@ Here we give an example of how to measure performance of TVM compiled models.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 2 minutes 24.044 seconds)
+ **Total running time of the script:** ( 2 minutes 23.771 seconds)
.. _sphx_glr_download_how_to_deploy_models_deploy_prequantized_tflite.py:
diff --git a/docs/_sources/how_to/deploy_models/deploy_quantized.rst.txt b/docs/_sources/how_to/deploy_models/deploy_quantized.rst.txt
index a74aeb30db..b46d88f397 100644
--- a/docs/_sources/how_to/deploy_models/deploy_quantized.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_quantized.rst.txt
@@ -253,7 +253,7 @@ We create a Relay VM to build and execute the model.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 39.176 seconds)
+ **Total running time of the script:** ( 1 minutes 34.524 seconds)
.. _sphx_glr_download_how_to_deploy_models_deploy_quantized.py:
diff --git a/docs/_sources/how_to/deploy_models/deploy_ssd_gluoncv.rst.txt b/docs/_sources/how_to/deploy_models/deploy_ssd_gluoncv.rst.txt
index 27c413600f..8232c3930d 100644
--- a/docs/_sources/how_to/deploy_models/deploy_ssd_gluoncv.rst.txt
+++ b/docs/_sources/how_to/deploy_models/deploy_ssd_gluoncv.rst.txt
@@ -166,7 +166,7 @@ Convert and compile model for CPU.
data: None
input_sym_arg_type = in_param.infer_type()[0]
Downloading /workspace/.mxnet/models/ssd_512_resnet50_v1_voc-9c8b225a.zip from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/ssd_512_resnet50_v1_voc-9c8b225a.zip...
-
0%| | 0/132723 [00:00<?, ?KB/s]
4%|4 | 5592/132723 [00:00<00:02, 55893.13KB/s]
10%|# | 13456/132723 [00:00<00:01, 69265.50KB/s]
15%|#5 | 20383/132723 [00:00<00:02, 45004.27KB/s]
21%|##1 | 28154/132723 [00:00<00:01, 54726.62KB/s]
27%|##7 | 35946/132723 [00:00<00:01, 61644.23KB/s]
33%|###3 | 43853/132723 [00:00<00:01, 66847.41KB/s]
39%|###9 | 51776/132723 [00:00<00:01, 70549.64KB/s]
45%|####4 | 59709/132723 [00:00<00:00, 73176.49KB/s]
51%|#####1 | 67814/132723 [00:01<00:00, 75534.23KB/s]
57%|#####7 | 75682/132723 [00:01<00:00, 76475.12KB/s]
63%|######3 | 83636/132723 [00:01<00:00, 77392.17KB/s]
69%|######8 | 91578/132723 [00:01<00:00, 77998.16KB/s]
75%|#######4 | 99468/132723 [00:01<00:00, 78267.40KB/s]
81%|######## | 107500/132723 [00:01<00:00, 78880.10KB/s]
87%|########6 | 115428/132723 [00:01<00:00, 78994.92KB/s]
93%|#########
2| 123430/132723 [00:01<00:00, 79295.94KB/s]
99%|#########9| 131416/132723 [00:01<00:00, 79463.16KB/s]
100%|##########| 132723/132723 [00:01<00:00, 72324.04KB/s]
+
0%| | 0/132723 [00:00<?, ?KB/s]
4%|4 | 5746/132723 [00:00<00:02, 57456.90KB/s]
10%|# | 13768/132723 [00:00<00:01, 70841.54KB/s]
17%|#6 | 21911/132723 [00:00<00:01, 75674.75KB/s]
23%|##2 | 30035/132723 [00:00<00:01, 77870.10KB/s]
29%|##8 | 38138/132723 [00:00<00:01, 79008.27KB/s]
35%|###4 | 46294/132723 [00:00<00:01, 79874.56KB/s]
41%|#### | 54394/132723 [00:00<00:00, 80239.38KB/s]
47%|####7 | 62472/132723 [00:00<00:00, 80407.54KB/s]
53%|#####3 | 70599/132723 [00:00<00:00, 80671.33KB/s]
59%|#####9 | 78742/132723 [00:01<00:00, 80898.22KB/s]
65%|######5 | 86832/132723 [00:01<00:00, 75357.26KB/s]
72%|#######1 | 94953/132723 [00:01<00:00, 77049.49KB/s]
78%|#######7 | 103065/132723 [00:01<00:00, 78238.09KB/s]
84%|########3 | 111235/132723 [00:01<00:00, 79253.13KB/s]
90%|########9 | 119361/132723 [00:01<00:00, 79844.53KB/s]
96%|########
#6| 127457/132723 [00:01<00:00, 80174.48KB/s]
100%|##########| 132723/132723 [00:01<00:00, 78566.50KB/s]
@@ -242,7 +242,7 @@ Display result
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 3 minutes 5.609 seconds)
+ **Total running time of the script:** ( 3 minutes 6.504 seconds)
.. _sphx_glr_download_how_to_deploy_models_deploy_ssd_gluoncv.py:
diff --git a/docs/_sources/how_to/deploy_models/sg_execution_times.rst.txt b/docs/_sources/how_to/deploy_models/sg_execution_times.rst.txt
index 6fdc5f321d..5ed011b00c 100644
--- a/docs/_sources/how_to/deploy_models/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/deploy_models/sg_execution_times.rst.txt
@@ -5,26 +5,26 @@
Computation times
=================
-**13:43.432** total execution time for **how_to_deploy_models** files:
+**13:45.186** total execution time for **how_to_deploy_models** files:
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_object_detection_pytorch.py` (``deploy_object_detection_pytorch.py``) | 03:12.782 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_object_detection_pytorch.py` (``deploy_object_detection_pytorch.py``) | 03:16.741 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_ssd_gluoncv.py` (``deploy_ssd_gluoncv.py``) | 03:05.609 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_ssd_gluoncv.py` (``deploy_ssd_gluoncv.py``) | 03:06.504 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_prequantized_tflite.py` (``deploy_prequantized_tflite.py``) | 02:24.044 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_prequantized_tflite.py` (``deploy_prequantized_tflite.py``) | 02:23.771 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_quantized.py` (``deploy_quantized.py``) | 01:39.176 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_quantized.py` (``deploy_quantized.py``) | 01:34.524 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_prequantized.py` (``deploy_prequantized.py``) | 01:06.105 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_prequantized.py` (``deploy_prequantized.py``) | 01:06.981 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_adreno.py` (``deploy_model_on_adreno.py``) | 00:51.040 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_adreno.py` (``deploy_model_on_adreno.py``) | 00:51.228 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_android.py` (``deploy_model_on_android.py``) | 00:35.265 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_android.py` (``deploy_model_on_android.py``) | 00:35.657 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_nano.py` (``deploy_model_on_nano.py``) | 00:24.929 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_nano.py` (``deploy_model_on_nano.py``) | 00:25.140 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_rasp.py` (``deploy_model_on_rasp.py``) | 00:24.475 | 0.0 MB |
+| :ref:`sphx_glr_how_to_deploy_models_deploy_model_on_rasp.py` (``deploy_model_on_rasp.py``) | 00:24.632 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_deploy_models_deploy_sparse.py` (``deploy_sparse.py``) | 00:00.007 | 0.0 MB |
+------------------------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/extend_tvm/bring_your_own_datatypes.rst.txt b/docs/_sources/how_to/extend_tvm/bring_your_own_datatypes.rst.txt
index 8e916a96ac..43c65ae8d6 100644
--- a/docs/_sources/how_to/extend_tvm/bring_your_own_datatypes.rst.txt
+++ b/docs/_sources/how_to/extend_tvm/bring_your_own_datatypes.rst.txt
@@ -472,7 +472,7 @@ First let us define two helper functions to get the mobilenet model and a cat im
.. code-block:: none
- Downloading /workspace/.mxnet/models/mobilenet0.25-9f83e440.zip519d4d3a-4bac-4731-9b50-35fb1592ae8e from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/mobilenet0.25-9f83e440.zip...
+ Downloading /workspace/.mxnet/models/mobilenet0.25-9f83e440.zip9f705891-4420-4dc7-8e6e-41d68c0ffa5e from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/mobilenet0.25-9f83e440.zip...
diff --git a/docs/_sources/how_to/extend_tvm/sg_execution_times.rst.txt b/docs/_sources/how_to/extend_tvm/sg_execution_times.rst.txt
index e12e9e1464..6c8fa746b6 100644
--- a/docs/_sources/how_to/extend_tvm/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/extend_tvm/sg_execution_times.rst.txt
@@ -5,14 +5,14 @@
Computation times
=================
-**00:47.168** total execution time for **how_to_extend_tvm** files:
+**00:47.320** total execution time for **how_to_extend_tvm** files:
+-------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_extend_tvm_bring_your_own_datatypes.py` (``bring_your_own_datatypes.py``) | 00:43.742 | 0.0 MB |
+| :ref:`sphx_glr_how_to_extend_tvm_bring_your_own_datatypes.py` (``bring_your_own_datatypes.py``) | 00:43.874 | 0.0 MB |
+-------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_extend_tvm_use_pass_instrument.py` (``use_pass_instrument.py``) | 00:02.398 | 0.0 MB |
+| :ref:`sphx_glr_how_to_extend_tvm_use_pass_instrument.py` (``use_pass_instrument.py``) | 00:02.412 | 0.0 MB |
+-------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_extend_tvm_use_pass_infra.py` (``use_pass_infra.py``) | 00:01.020 | 0.0 MB |
+| :ref:`sphx_glr_how_to_extend_tvm_use_pass_infra.py` (``use_pass_infra.py``) | 00:01.026 | 0.0 MB |
+-------------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_extend_tvm_low_level_custom_pass.py` (``low_level_custom_pass.py``) | 00:00.007 | 0.0 MB |
+-------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/extend_tvm/use_pass_instrument.rst.txt b/docs/_sources/how_to/extend_tvm/use_pass_instrument.rst.txt
index a6a41f60aa..6f869c1744 100644
--- a/docs/_sources/how_to/extend_tvm/use_pass_instrument.rst.txt
+++ b/docs/_sources/how_to/extend_tvm/use_pass_instrument.rst.txt
@@ -216,10 +216,10 @@ profile the execution time of each passes.
.. code-block:: none
Printing results of timing profile...
- InferType: 7117us [7117us] (46.52%; 46.52%)
- FoldScaleAxis: 8182us [7us] (53.48%; 53.48%)
- FoldConstant: 8176us [1664us] (53.44%; 99.92%)
- InferType: 6511us [6511us] (42.56%; 79.64%)
+ InferType: 7125us [7125us] (46.15%; 46.15%)
+ FoldScaleAxis: 8313us [6us] (53.85%; 53.85%)
+ FoldConstant: 8307us [1681us] (53.81%; 99.93%)
+ InferType: 6626us [6626us] (42.92%; 79.76%)
@@ -258,10 +258,10 @@ Refer to following sections and :py:func:`tvm.instrument.pass_instrument` for th
.. code-block:: none
Printing results of timing profile...
- InferType: 6612us [6612us] (45.22%; 45.22%)
- FoldScaleAxis: 8010us [5us] (54.78%; 54.78%)
- FoldConstant: 8005us [1642us] (54.75%; 99.94%)
- InferType: 6363us [6363us] (43.52%; 79.49%)
+ InferType: 6724us [6724us] (44.23%; 44.23%)
+ FoldScaleAxis: 8479us [6us] (55.77%; 55.77%)
+ FoldConstant: 8473us [1715us] (55.73%; 99.93%)
+ InferType: 6758us [6758us] (44.45%; 79.76%)
diff --git a/docs/_sources/how_to/optimize_operators/opt_conv_cuda.rst.txt b/docs/_sources/how_to/optimize_operators/opt_conv_cuda.rst.txt
index d24ea13d46..9c50ff32ca 100644
--- a/docs/_sources/how_to/optimize_operators/opt_conv_cuda.rst.txt
+++ b/docs/_sources/how_to/optimize_operators/opt_conv_cuda.rst.txt
@@ -340,7 +340,7 @@ latency of convolution.
.. code-block:: none
- Convolution: 54.214591 ms
+ Convolution: 46.378784 ms
diff --git a/docs/_sources/how_to/optimize_operators/opt_conv_tensorcore.rst.txt b/docs/_sources/how_to/optimize_operators/opt_conv_tensorcore.rst.txt
index 947e57f599..8b9a0df279 100644
--- a/docs/_sources/how_to/optimize_operators/opt_conv_tensorcore.rst.txt
+++ b/docs/_sources/how_to/optimize_operators/opt_conv_tensorcore.rst.txt
@@ -657,7 +657,7 @@ be able to run on our build server
.. code-block:: none
- conv2d with tensor core: 13.313641 ms
+ conv2d with tensor core: 13.351321 ms
diff --git a/docs/_sources/how_to/optimize_operators/opt_gemm.rst.txt b/docs/_sources/how_to/optimize_operators/opt_gemm.rst.txt
index 70122d27c9..9b10961d8f 100644
--- a/docs/_sources/how_to/optimize_operators/opt_gemm.rst.txt
+++ b/docs/_sources/how_to/optimize_operators/opt_gemm.rst.txt
@@ -143,8 +143,8 @@ Then we write a baseline implementation, the simplest way to write a matrix mult
.. code-block:: none
- Numpy running time: 0.018498
- Baseline: 3.376149
+ Numpy running time: 0.018197
+ Baseline: 3.276830
@@ -238,7 +238,7 @@ fill 32 * 32 * sizeof(float) which is 4KB in the cache whose total size is 32KB
.. code-block:: none
- Opt1: 0.291363
+ Opt1: 0.303195
@@ -340,7 +340,7 @@ In this tutorial, we chose to vectorize the inner loop row data since it is cach
.. code-block:: none
- Opt2: 0.329749
+ Opt2: 0.336533
@@ -435,7 +435,7 @@ the access pattern for A matrix is more cache friendly.
.. code-block:: none
- Opt3: 0.115922
+ Opt3: 0.120270
@@ -559,7 +559,7 @@ flattening.
.. code-block:: none
- Opt4: 0.109779
+ Opt4: 0.110110
@@ -680,7 +680,7 @@ write to C when all the block results are ready.
.. code-block:: none
- Opt5: 0.111204
+ Opt5: 0.110684
@@ -804,7 +804,7 @@ Furthermore, we can also utilize multi-core processors to do the thread-level pa
.. code-block:: none
- Opt6: 0.146807
+ Opt6: 0.147054
diff --git a/docs/_sources/how_to/optimize_operators/sg_execution_times.rst.txt b/docs/_sources/how_to/optimize_operators/sg_execution_times.rst.txt
index 3918317fdc..d1bd3fe4fa 100644
--- a/docs/_sources/how_to/optimize_operators/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/optimize_operators/sg_execution_times.rst.txt
@@ -5,12 +5,12 @@
Computation times
=================
-**00:34.560** total execution time for **how_to_optimize_operators** files:
+**00:34.533** total execution time for **how_to_optimize_operators** files:
+-----------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_optimize_operators_opt_gemm.py` (``opt_gemm.py``) | 00:32.022 | 0.0 MB |
+| :ref:`sphx_glr_how_to_optimize_operators_opt_gemm.py` (``opt_gemm.py``) | 00:31.944 | 0.0 MB |
+-----------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_optimize_operators_opt_conv_tensorcore.py` (``opt_conv_tensorcore.py``) | 00:01.495 | 0.0 MB |
+| :ref:`sphx_glr_how_to_optimize_operators_opt_conv_tensorcore.py` (``opt_conv_tensorcore.py``) | 00:01.519 | 0.0 MB |
+-----------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_optimize_operators_opt_conv_cuda.py` (``opt_conv_cuda.py``) | 00:01.044 | 0.0 MB |
+| :ref:`sphx_glr_how_to_optimize_operators_opt_conv_cuda.py` (``opt_conv_cuda.py``) | 00:01.069 | 0.0 MB |
+-----------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/tune_with_autoscheduler/sg_execution_times.rst.txt b/docs/_sources/how_to/tune_with_autoscheduler/sg_execution_times.rst.txt
index 87eba4c516..6e98310d1b 100644
--- a/docs/_sources/how_to/tune_with_autoscheduler/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/tune_with_autoscheduler/sg_execution_times.rst.txt
@@ -5,18 +5,18 @@
Computation times
=================
-**09:05.595** total execution time for **how_to_tune_with_autoscheduler** files:
+**08:51.610** total execution time for **how_to_tune_with_autoscheduler** files:
+----------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_conv2d_layer_cuda.py` (``tune_conv2d_layer_cuda.py``) | 05:42.541 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_conv2d_layer_cuda.py` (``tune_conv2d_layer_cuda.py``) | 05:27.158 | 0.0 MB |
+----------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_x86.py` (``tune_network_x86.py``) | 01:31.335 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_x86.py` (``tune_network_x86.py``) | 01:31.258 | 0.0 MB |
+----------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_cuda.py` (``tune_network_cuda.py``) | 01:01.296 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_cuda.py` (``tune_network_cuda.py``) | 01:01.762 | 0.0 MB |
+----------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_sparse_x86.py` (``tune_sparse_x86.py``) | 00:27.353 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_sparse_x86.py` (``tune_sparse_x86.py``) | 00:28.295 | 0.0 MB |
+----------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_arm.py` (``tune_network_arm.py``) | 00:12.000 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_arm.py` (``tune_network_arm.py``) | 00:11.940 | 0.0 MB |
+----------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_mali.py` (``tune_network_mali.py``) | 00:11.071 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autoscheduler_tune_network_mali.py` (``tune_network_mali.py``) | 00:11.198 | 0.0 MB |
+----------------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.rst.txt b/docs/_sources/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.rst.txt
index 2a957d761a..b0b8fc88ee 100644
--- a/docs/_sources/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.rst.txt
+++ b/docs/_sources/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.rst.txt
@@ -239,1282 +239,125 @@ cooperative fetching, unrolling and operator fusion.
bias: Buffer(bias_2: Pointer(float32), float32, [1, 512, 1, 1], []),
compute: Buffer(compute_2: Pointer(float32), float32, [1, 512, 7, 7], [])}
buffer_map = {data_1: data, kernel_1: kernel, bias_1: bias, compute_1: compute} {
- attr [IterVar(blockIdx.x: int32, (nullptr), "ThreadIndex", "blockIdx.x")] "thread_extent" = 16;
- allocate(conv2d_nchw: Pointer(local float32), float32, [28]), storage_scope = local;
- allocate(pad_temp.shared: Pointer(shared float32), float32, [1296]), storage_scope = shared;
- allocate(kernel.shared: Pointer(shared float32), float32, [4608]), storage_scope = shared;
- attr [IterVar(threadIdx.x: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56 {
- conv2d_nchw_1: Buffer(conv2d_nchw, float32, [196], [], scope="local", align=32)[0] = 0f32
- conv2d_nchw_1[14] = 0f32
+ attr [IterVar(blockIdx.x: int32, (nullptr), "ThreadIndex", "blockIdx.x")] "thread_extent" = 28;
+ allocate(conv2d_nchw: Pointer(local float32), float32, [4]), storage_scope = local;
+ allocate(pad_temp.shared: Pointer(shared float32), float32, [144]), storage_scope = shared;
+ allocate(kernel.shared: Pointer(shared float32), float32, [6144]), storage_scope = shared;
+ attr [IterVar(threadIdx.x: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224 {
+ conv2d_nchw_1: Buffer(conv2d_nchw, float32, [1], [], scope="local", align=4)[0] = 0f32
conv2d_nchw_1[1] = 0f32
- conv2d_nchw_1[15] = 0f32
conv2d_nchw_1[2] = 0f32
- conv2d_nchw_1[16] = 0f32
conv2d_nchw_1[3] = 0f32
- conv2d_nchw_1[17] = 0f32
- conv2d_nchw_1[4] = 0f32
- conv2d_nchw_1[18] = 0f32
- conv2d_nchw_1[5] = 0f32
- conv2d_nchw_1[19] = 0f32
- conv2d_nchw_1[6] = 0f32
- conv2d_nchw_1[20] = 0f32
- conv2d_nchw_1[7] = 0f32
- conv2d_nchw_1[21] = 0f32
- conv2d_nchw_1[8] = 0f32
- conv2d_nchw_1[22] = 0f32
- conv2d_nchw_1[9] = 0f32
- conv2d_nchw_1[23] = 0f32
- conv2d_nchw_1[10] = 0f32
- conv2d_nchw_1[24] = 0f32
- conv2d_nchw_1[11] = 0f32
- conv2d_nchw_1[25] = 0f32
- conv2d_nchw_1[12] = 0f32
- conv2d_nchw_1[26] = 0f32
- conv2d_nchw_1[13] = 0f32
- conv2d_nchw_1[27] = 0f32
for (rc.outer.outer: int32, 0, 32) {
- let cse_var_2: int32 = (rc.outer.outer*784)
- let cse_var_1: int32 = (rc.outer.outer*144)
- {
- attr [IterVar(threadIdx.x_1: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1: Buffer(pad_temp.shared, float32, [1296], [], scope="shared")[threadIdx.x_1] = @tir.if_then_else((((9 <= threadIdx.x_1) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3: Buffer(data_2, float32, [25088], [])[(((cse_var_2 + (floordiv(threadIdx.x_1, 9)*7)) + floormod(threadIdx.x_1, 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 56)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 56), 81)) && (floormod((threadIdx.x_1 + 56), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 2), 9))) && (floormod((threadIdx.x_1 + 2), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 56), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 56), 81), 9)*7)) + floormod((threadIdx.x_1 + 2), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 112)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 31), 81)) && (floormod((threadIdx.x_1 + 31), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 4), 9))) && (floormod((threadIdx.x_1 + 4), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 112), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 31), 81), 9)*7)) + floormod((threadIdx.x_1 + 4), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 168)] = @tir.if_then_else((((9 <= floormod((threadIdx.x_1 + 6), 81)) && (1 <= floormod((threadIdx.x_1 + 6), 9))) && (floormod((threadIdx.x_1 + 6), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 168), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 6), 81), 9)*7)) + floormod((threadIdx.x_1 + 6), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 224)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 62), 81)) && (floormod((threadIdx.x_1 + 62), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 8), 9))) && (floormod((threadIdx.x_1 + 8), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 224), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 62), 81), 9)*7)) + floormod((threadIdx.x_1 + 8), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 280)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 37), 81)) && (floormod((threadIdx.x_1 + 37), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 1), 9))) && (floormod((threadIdx.x_1 + 1), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 280), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 37), 81), 9)*7)) + floormod((threadIdx.x_1 + 1), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 336)] = @tir.if_then_else(((1 <= floormod((threadIdx.x_1 + 3), 9)) && (floormod((threadIdx.x_1 + 3), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 336), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 12), 81), 9)*7)) + floormod((threadIdx.x_1 + 3), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 392)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 68), 81)) && (floormod((threadIdx.x_1 + 68), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 5), 9))) && (floormod((threadIdx.x_1 + 5), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 392), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 68), 81), 9)*7)) + floormod((threadIdx.x_1 + 5), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 448)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 43), 81)) && (floormod((threadIdx.x_1 + 43), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 7), 9))) && (floormod((threadIdx.x_1 + 7), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 448), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 43), 81), 9)*7)) + floormod((threadIdx.x_1 + 7), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 504)] = @tir.if_then_else((((threadIdx.x_1 < 54) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 504), 81)*49)) + ((floordiv(threadIdx.x_1, 9) + 2)*7)) + floormod(threadIdx.x_1, 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 560)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 74), 81)) && (floormod((threadIdx.x_1 + 74), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 2), 9))) && (floormod((threadIdx.x_1 + 2), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 560), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 74), 81), 9)*7)) + floormod((threadIdx.x_1 + 2), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 616)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 49), 81)) && (floormod((threadIdx.x_1 + 49), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 4), 9))) && (floormod((threadIdx.x_1 + 4), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 616), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 49), 81), 9)*7)) + floormod((threadIdx.x_1 + 4), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 672)] = @tir.if_then_else((((threadIdx.x_1 < 48) && (1 <= floormod((threadIdx.x_1 + 6), 9))) && (floormod((threadIdx.x_1 + 6), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 672), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 24), 81), 9)*7)) + floormod((threadIdx.x_1 + 6), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 728)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 80), 81)) && (floormod((threadIdx.x_1 + 80), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 8), 9))) && (floormod((threadIdx.x_1 + 8), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 728), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 80), 81), 9)*7)) + floormod((threadIdx.x_1 + 8), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 784)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 55), 81)) && (floormod((threadIdx.x_1 + 55), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 1), 9))) && (floormod((threadIdx.x_1 + 1), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 784), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 55), 81), 9)*7)) + floormod((threadIdx.x_1 + 1), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 840)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 30), 81)) && (floormod((threadIdx.x_1 + 30), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 3), 9))) && (floormod((threadIdx.x_1 + 3), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 840), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 30), 81), 9)*7)) + floormod((threadIdx.x_1 + 3), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 896)] = @tir.if_then_else((((9 <= floormod((threadIdx.x_1 + 5), 81)) && (1 <= floormod((threadIdx.x_1 + 5), 9))) && (floormod((threadIdx.x_1 + 5), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 896), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 5), 81), 9)*7)) + floormod((threadIdx.x_1 + 5), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 952)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 61), 81)) && (floormod((threadIdx.x_1 + 61), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 7), 9))) && (floormod((threadIdx.x_1 + 7), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 952), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 61), 81), 9)*7)) + floormod((threadIdx.x_1 + 7), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1008)] = @tir.if_then_else(((((1 <= floormod((floordiv(threadIdx.x_1, 9) + 4), 9)) && (floormod((threadIdx.x_1 + 36), 81) < 72)) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1008), 81)*49)) + (floormod((floordiv(threadIdx.x_1, 9) + 4), 9)*7)) + floormod(threadIdx.x_1, 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1064)] = @tir.if_then_else(((1 <= floormod((threadIdx.x_1 + 2), 9)) && (floormod((threadIdx.x_1 + 2), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1064), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 11), 81), 9)*7)) + floormod((threadIdx.x_1 + 2), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1120)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 67), 81)) && (floormod((threadIdx.x_1 + 67), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 4), 9))) && (floormod((threadIdx.x_1 + 4), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1120), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 67), 81), 9)*7)) + floormod((threadIdx.x_1 + 4), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1176)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 42), 81)) && (floormod((threadIdx.x_1 + 42), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 6), 9))) && (floormod((threadIdx.x_1 + 6), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1176), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 42), 81), 9)*7)) + floormod((threadIdx.x_1 + 6), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1232)] = @tir.if_then_else((((threadIdx.x_1 < 55) && (1 <= floormod((threadIdx.x_1 + 8), 9))) && (floormod((threadIdx.x_1 + 8), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1232), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 17), 81), 9)*7)) + floormod((threadIdx.x_1 + 8), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- if @tir.likely((threadIdx.x_1 < 8), dtype=bool) {
- pad_temp.shared_1[(threadIdx.x_1 + 1288)] = 0f32
- }
- attr [IterVar(threadIdx.x_2: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1: Buffer(kernel.shared, float32, [4608], [], scope="shared")[threadIdx.x_2] = kernel_3: Buffer(kernel_2, float32, [2359296], [])[(((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 56)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + (floordiv((threadIdx.x_2 + 56), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 112)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 112), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 168)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 168), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 224)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 224), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 280)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 280), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 336)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 336), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 392)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 392), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 448)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 448), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 504)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 504), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 560)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 560), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 616)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 616), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 672)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 672), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 728)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 728), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 784)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 784), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 840)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 840), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 896)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 896), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 952)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 952), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1008)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 32256)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1064)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1064), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1120)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1120), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1176)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1176), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1232)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1232), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1288)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1288), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1344)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1344), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1400)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1400), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1456)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1456), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1512)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1512), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1568)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1568), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1624)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1624), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1680)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1680), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1736)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1736), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1792)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1792), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1848)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1848), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1904)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1904), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1960)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1960), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2016)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 64512)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2072)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2072), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2128)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2128), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2184)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2184), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2240)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2240), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2296)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2296), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2352)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2352), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2408)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2408), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2464)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2464), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2520)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2520), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2576)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2576), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2632)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2632), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2688)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2688), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2744)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2744), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2800)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2800), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2856)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2856), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2912)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2912), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2968)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2968), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3024)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 96768)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3080)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3080), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3136)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3136), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3192)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3192), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3248)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3248), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3304)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3304), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3360)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3360), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3416)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3416), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3472)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3472), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3528)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3528), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3584)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3584), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3640)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3640), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3696)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3696), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3752)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3752), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3808)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3808), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3864)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3864), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3920)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3920), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3976)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3976), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4032)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 129024)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4088)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4088), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4144)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4144), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4200)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4200), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4256)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4256), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4312)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4312), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4368)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4368), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4424)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4424), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4480)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4480), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4536)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4536), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- if @tir.likely((threadIdx.x_2 < 16), dtype=bool) {
- kernel.shared_1[(threadIdx.x_2 + 4592)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4592), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- }
- for (rc.outer.inner: int32, 0, 4) {
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- }
- }
- }
- for (i1.inner: int32, 0, 2) {
- for (i3.inner: int32, 0, 7) {
- let cse_var_3: int32 = ((i1.inner*7) + i3.inner)
+ for (ry.outer.outer: int32, 0, 3) {
+ let cse_var_2: int32 = (rc.outer.outer*144)
+ let cse_var_1: int32 = (ry.outer.outer*3)
{
- compute_3: Buffer(compute_2, float32, [25088], [])[(((((blockIdx.x*1568) + (floordiv(threadIdx.x, 7)*98)) + (i1.inner*49)) + (floormod(threadIdx.x, 7)*7)) + i3.inner)] = max((conv2d_nchw_1[cse_var_3] + bias_3: Buffer(bias_2, float32, [512], [])[(((blockIdx.x*32) + (floordiv(threadIdx.x, 7)*2)) + i1.inner)]), 0f32)
- compute_3[((((((blockIdx.x*1568) + (floordiv(threadIdx.x, 7)*98)) + (i1.inner*49)) + (floormod(threadIdx.x, 7)*7)) + i3.inner) + 784)] = max((conv2d_nchw_1[(cse_var_3 + 14)] + bias_3[((((blockIdx.x*32) + (floordiv(threadIdx.x, 7)*2)) + i1.inner) + 16)]), 0f32)
+ attr [IterVar(threadIdx.x_1: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ if @tir.likely((threadIdx.x_1 < 144), dtype=bool) {
+ pad_temp.shared_1: Buffer(pad_temp.shared, float32, [144], [], scope="shared")[threadIdx.x_1] = @tir.if_then_else(((((1 <= (ry.outer.outer + floormod(blockIdx.x, 7))) && ((ry.outer.outer + floormod(blockIdx.x, 7)) < 8)) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3: Buffer(data_2, float32, [25088], [])[((((((rc.outer.outer*784) + (floordiv(threadIdx.x_1, 9)*49)) + (ry.outer.outer*7)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx. [...]
+ }
+ attr [IterVar(threadIdx.x_2: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1: Buffer(kernel.shared, float32, [6144], [], scope="shared")[threadIdx.x_2] = kernel_3: Buffer(kernel_2, float32, [2359296], [])[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 224)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 224), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 448)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 448), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 672)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 64512)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 896)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 896), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1120)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 1120), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1344)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 129024)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1568)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 1568), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1792)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 1792), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2016)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 193536)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2240)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 2240), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2464)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 2464), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2688)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 258048)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2912)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 2912), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3136)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 3136), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3360)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 322560)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3584)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 3584), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3808)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 3808), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4032)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 387072)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4256)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 4256), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4480)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 4480), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4704)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 451584)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4928)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 4928), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5152)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 5152), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5376)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 516096)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5600)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 5600), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5824)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 5824), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ if @tir.likely((threadIdx.x_2 < 96), dtype=bool) {
+ kernel.shared_1[(threadIdx.x_2 + 6048)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 580608)]
+ }
+ for (rc.outer.inner: int32, 0, 2) {
+ for (rx.outer.inner: int32, 0, 3) {
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1536)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3072)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4608)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1539)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3075)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4611)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 6)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1542)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3078)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4614)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 9)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1545)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3081)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4617)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 12)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1548)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3084)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4620)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 15)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1551)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3087)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4623)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 18)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1554)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3090)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4626)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 21)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1557)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3093)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4629)]))
+ }
+ }
}
}
}
+ compute_3: Buffer(compute_2, float32, [25088], [])[((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7))] = max((conv2d_nchw_1[0] + bias_3: Buffer(bias_2, float32, [512], [])[((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7))]), 0f32)
+ compute_3[(((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7)) + 1568)] = max((conv2d_nchw_1[1] + bias_3[(((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7)) + 32)]), 0f32)
+ compute_3[(((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7)) + 3136)] = max((conv2d_nchw_1[2] + bias_3[(((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7)) + 64)]), 0f32)
+ compute_3[(((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7)) + 4704)] = max((conv2d_nchw_1[3] + bias_3[(((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7)) + 96)]), 0f32)
}
}
@@ -1568,7 +411,7 @@ We build the binary and check its correctness and performance.
.. code-block:: none
- Execution time of this operator: 0.268 ms
+ Execution time of this operator: 0.387 ms
@@ -1617,35 +460,35 @@ They can be used for debugging and learning the behavior of the auto-scheduler.
conv2d_nchw_nn_o_o_o_i, conv2d_nchw_nn_o_o_i = s[conv2d_nchw].split(conv2d_nchw_nn_o_o_i, factor=1)
conv2d_nchw_nn_o_o_o_o, conv2d_nchw_nn_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_nn_o_o_o_i, factor=1)
conv2d_nchw_ff_o_i, conv2d_nchw_ff_i = s[conv2d_nchw].split(conv2d_nchw_ff, factor=1)
- conv2d_nchw_ff_o_o_i, conv2d_nchw_ff_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_i, factor=2)
- conv2d_nchw_ff_o_o_o_i, conv2d_nchw_ff_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_i, factor=8)
- conv2d_nchw_ff_o_o_o_o, conv2d_nchw_ff_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_o_i, factor=2)
+ conv2d_nchw_ff_o_o_i, conv2d_nchw_ff_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_i, factor=1)
+ conv2d_nchw_ff_o_o_o_i, conv2d_nchw_ff_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_i, factor=32)
+ conv2d_nchw_ff_o_o_o_o, conv2d_nchw_ff_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_o_i, factor=4)
conv2d_nchw_yy_o_i, conv2d_nchw_yy_i = s[conv2d_nchw].split(conv2d_nchw_yy, factor=1)
conv2d_nchw_yy_o_o_i, conv2d_nchw_yy_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_i, factor=1)
- conv2d_nchw_yy_o_o_o_i, conv2d_nchw_yy_o_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_o_i, factor=7)
+ conv2d_nchw_yy_o_o_o_i, conv2d_nchw_yy_o_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_o_i, factor=1)
conv2d_nchw_yy_o_o_o_o, conv2d_nchw_yy_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_o_o_i, factor=1)
- conv2d_nchw_xx_o_i, conv2d_nchw_xx_i = s[conv2d_nchw].split(conv2d_nchw_xx, factor=7)
+ conv2d_nchw_xx_o_i, conv2d_nchw_xx_i = s[conv2d_nchw].split(conv2d_nchw_xx, factor=1)
conv2d_nchw_xx_o_o_i, conv2d_nchw_xx_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_i, factor=1)
- conv2d_nchw_xx_o_o_o_i, conv2d_nchw_xx_o_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_o_i, factor=1)
+ conv2d_nchw_xx_o_o_o_i, conv2d_nchw_xx_o_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_o_i, factor=7)
conv2d_nchw_xx_o_o_o_o, conv2d_nchw_xx_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_o_o_i, factor=1)
- conv2d_nchw_rc_o_i, conv2d_nchw_rc_i = s[conv2d_nchw].split(conv2d_nchw_rc, factor=4)
- conv2d_nchw_rc_o_o, conv2d_nchw_rc_o_i = s[conv2d_nchw].split(conv2d_nchw_rc_o_i, factor=4)
- conv2d_nchw_ry_o_i, conv2d_nchw_ry_i = s[conv2d_nchw].split(conv2d_nchw_ry, factor=3)
+ conv2d_nchw_rc_o_i, conv2d_nchw_rc_i = s[conv2d_nchw].split(conv2d_nchw_rc, factor=8)
+ conv2d_nchw_rc_o_o, conv2d_nchw_rc_o_i = s[conv2d_nchw].split(conv2d_nchw_rc_o_i, factor=2)
+ conv2d_nchw_ry_o_i, conv2d_nchw_ry_i = s[conv2d_nchw].split(conv2d_nchw_ry, factor=1)
conv2d_nchw_ry_o_o, conv2d_nchw_ry_o_i = s[conv2d_nchw].split(conv2d_nchw_ry_o_i, factor=1)
- conv2d_nchw_rx_o_i, conv2d_nchw_rx_i = s[conv2d_nchw].split(conv2d_nchw_rx, factor=3)
- conv2d_nchw_rx_o_o, conv2d_nchw_rx_o_i = s[conv2d_nchw].split(conv2d_nchw_rx_o_i, factor=1)
+ conv2d_nchw_rx_o_i, conv2d_nchw_rx_i = s[conv2d_nchw].split(conv2d_nchw_rx, factor=1)
+ conv2d_nchw_rx_o_o, conv2d_nchw_rx_o_i = s[conv2d_nchw].split(conv2d_nchw_rx_o_i, factor=3)
s[conv2d_nchw].reorder(conv2d_nchw_nn_o_o_o_o, conv2d_nchw_ff_o_o_o_o, conv2d_nchw_yy_o_o_o_o, conv2d_nchw_xx_o_o_o_o, conv2d_nchw_nn_o_o_o_i, conv2d_nchw_ff_o_o_o_i, conv2d_nchw_yy_o_o_o_i, conv2d_nchw_xx_o_o_o_i, conv2d_nchw_nn_o_o_i, conv2d_nchw_ff_o_o_i, conv2d_nchw_yy_o_o_i, conv2d_nchw_xx_o_o_i, conv2d_nchw_rc_o_o, conv2d_nchw_ry_o_o, conv2d_nchw_rx_o_o, conv2d_nchw_rc_o_i, conv2d_nchw_ry_o_i, conv2d_nchw_rx_o_i, conv2d_nchw_nn_o_i, conv2d_nchw_ff_o_i, conv2d_nchw_yy_o_i, conv2 [...]
compute_i0_o_i, compute_i0_i = s[compute].split(compute_i0, factor=1)
compute_i0_o_o_i, compute_i0_o_i = s[compute].split(compute_i0_o_i, factor=1)
compute_i0_o_o_o, compute_i0_o_o_i = s[compute].split(compute_i0_o_o_i, factor=1)
- compute_i1_o_i, compute_i1_i = s[compute].split(compute_i1, factor=2)
- compute_i1_o_o_i, compute_i1_o_i = s[compute].split(compute_i1_o_i, factor=8)
- compute_i1_o_o_o, compute_i1_o_o_i = s[compute].split(compute_i1_o_o_i, factor=2)
+ compute_i1_o_i, compute_i1_i = s[compute].split(compute_i1, factor=1)
+ compute_i1_o_o_i, compute_i1_o_i = s[compute].split(compute_i1_o_i, factor=32)
+ compute_i1_o_o_o, compute_i1_o_o_i = s[compute].split(compute_i1_o_o_i, factor=4)
compute_i2_o_i, compute_i2_i = s[compute].split(compute_i2, factor=1)
- compute_i2_o_o_i, compute_i2_o_i = s[compute].split(compute_i2_o_i, factor=7)
+ compute_i2_o_o_i, compute_i2_o_i = s[compute].split(compute_i2_o_i, factor=1)
compute_i2_o_o_o, compute_i2_o_o_i = s[compute].split(compute_i2_o_o_i, factor=1)
- compute_i3_o_i, compute_i3_i = s[compute].split(compute_i3, factor=7)
- compute_i3_o_o_i, compute_i3_o_i = s[compute].split(compute_i3_o_i, factor=1)
+ compute_i3_o_i, compute_i3_i = s[compute].split(compute_i3, factor=1)
+ compute_i3_o_o_i, compute_i3_o_i = s[compute].split(compute_i3_o_i, factor=7)
compute_i3_o_o_o, compute_i3_o_o_i = s[compute].split(compute_i3_o_o_i, factor=1)
s[compute].reorder(compute_i0_o_o_o, compute_i1_o_o_o, compute_i2_o_o_o, compute_i3_o_o_o, compute_i0_o_o_i, compute_i1_o_o_i, compute_i2_o_o_i, compute_i3_o_o_i, compute_i0_o_i, compute_i1_o_i, compute_i2_o_i, compute_i3_o_i, compute_i0_i, compute_i1_i, compute_i2_i, compute_i3_i)
s[conv2d_nchw].compute_at(s[compute], compute_i3_o_i)
@@ -1665,14 +508,14 @@ They can be used for debugging and learning the behavior of the auto-scheduler.
kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused = s[kernel_shared].fuse(kernel_shared_ax0, kernel_shared_ax1, kernel_shared_ax2, kernel_shared_ax3)
kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i = s[kernel_shared].split(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused, factor=1)
s[kernel_shared].vectorize(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i)
- kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[kernel_shared].split(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=56)
+ kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[kernel_shared].split(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=224)
s[kernel_shared].bind(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i, te.thread_axis("threadIdx.x"))
pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused = s[pad_temp_shared].fuse(pad_temp_shared_ax0, pad_temp_shared_ax1, pad_temp_shared_ax2, pad_temp_shared_ax3)
pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i = s[pad_temp_shared].split(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused, factor=1)
s[pad_temp_shared].vectorize(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i)
- pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[pad_temp_shared].split(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=56)
+ pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[pad_temp_shared].split(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=224)
s[pad_temp_shared].bind(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i, te.thread_axis("threadIdx.x"))
- s[conv2d_nchw].pragma(conv2d_nchw_nn_o_o_o_o, "auto_unroll_max_step", 1024)
+ s[conv2d_nchw].pragma(conv2d_nchw_nn_o_o_o_o, "auto_unroll_max_step", 64)
s[conv2d_nchw].pragma(conv2d_nchw_nn_o_o_o_o, "unroll_explicit", True)
CUDA source code:
@@ -1690,1169 +533,93 @@ They can be used for debugging and learning the behavior of the auto-scheduler.
#define int64_t long long
#define uint64_t unsigned long long
#endif
- extern "C" __global__ void __launch_bounds__(56) default_function_kernel0(float* __restrict__ data, float* __restrict__ kernel, float* __restrict__ compute, float* __restrict__ bias) {
- float conv2d_nchw[28];
- __shared__ float pad_temp_shared[1296];
- __shared__ float kernel_shared[4608];
+ extern "C" __global__ void __launch_bounds__(224) default_function_kernel0(float* __restrict__ data, float* __restrict__ kernel, float* __restrict__ compute, float* __restrict__ bias) {
+ float conv2d_nchw[4];
+ __shared__ float pad_temp_shared[144];
+ __shared__ float kernel_shared[6144];
conv2d_nchw[0] = 0.000000e+00f;
- conv2d_nchw[14] = 0.000000e+00f;
conv2d_nchw[1] = 0.000000e+00f;
- conv2d_nchw[15] = 0.000000e+00f;
conv2d_nchw[2] = 0.000000e+00f;
- conv2d_nchw[16] = 0.000000e+00f;
conv2d_nchw[3] = 0.000000e+00f;
- conv2d_nchw[17] = 0.000000e+00f;
- conv2d_nchw[4] = 0.000000e+00f;
- conv2d_nchw[18] = 0.000000e+00f;
- conv2d_nchw[5] = 0.000000e+00f;
- conv2d_nchw[19] = 0.000000e+00f;
- conv2d_nchw[6] = 0.000000e+00f;
- conv2d_nchw[20] = 0.000000e+00f;
- conv2d_nchw[7] = 0.000000e+00f;
- conv2d_nchw[21] = 0.000000e+00f;
- conv2d_nchw[8] = 0.000000e+00f;
- conv2d_nchw[22] = 0.000000e+00f;
- conv2d_nchw[9] = 0.000000e+00f;
- conv2d_nchw[23] = 0.000000e+00f;
- conv2d_nchw[10] = 0.000000e+00f;
- conv2d_nchw[24] = 0.000000e+00f;
- conv2d_nchw[11] = 0.000000e+00f;
- conv2d_nchw[25] = 0.000000e+00f;
- conv2d_nchw[12] = 0.000000e+00f;
- conv2d_nchw[26] = 0.000000e+00f;
- conv2d_nchw[13] = 0.000000e+00f;
- conv2d_nchw[27] = 0.000000e+00f;
for (int rc_outer_outer = 0; rc_outer_outer < 32; ++rc_outer_outer) {
- __syncthreads();
- pad_temp_shared[((int)threadIdx.x)] = ((((9 <= ((int)threadIdx.x)) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[((((rc_outer_outer * 784) + ((((int)threadIdx.x) / 9) * 7)) + (((int)threadIdx.x) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 56)] = (((((9 <= ((((int)threadIdx.x) + 56) % 81)) && (((((int)threadIdx.x) + 56) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 2) % 9))) && (((((int)threadIdx.x) + 2) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 56) / 81) * 49)) + ((((((int)threadIdx.x) + 56) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 2) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 112)] = (((((9 <= ((((int)threadIdx.x) + 31) % 81)) && (((((int)threadIdx.x) + 31) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 4) % 9))) && (((((int)threadIdx.x) + 4) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 112) / 81) * 49)) + ((((((int)threadIdx.x) + 31) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 4) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 168)] = ((((3 <= ((int)threadIdx.x)) && (1 <= ((((int)threadIdx.x) + 6) % 9))) && (((((int)threadIdx.x) + 6) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 168) / 81) * 49)) + (((((int)threadIdx.x) + 6) / 9) * 7)) + ((((int)threadIdx.x) + 6) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 224)] = (((((9 <= ((((int)threadIdx.x) + 62) % 81)) && (((((int)threadIdx.x) + 62) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 8) % 9))) && (((((int)threadIdx.x) + 8) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 224) / 81) * 49)) + ((((((int)threadIdx.x) + 62) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 8) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 280)] = (((((9 <= ((((int)threadIdx.x) + 37) % 81)) && (((((int)threadIdx.x) + 37) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 1) % 9))) && (((((int)threadIdx.x) + 1) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 280) / 81) * 49)) + ((((((int)threadIdx.x) + 37) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 1) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 336)] = (((1 <= ((((int)threadIdx.x) + 3) % 9)) && (((((int)threadIdx.x) + 3) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 336) / 81) * 49)) + (((((int)threadIdx.x) + 12) / 9) * 7)) + ((((int)threadIdx.x) + 3) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 392)] = (((((9 <= ((((int)threadIdx.x) + 68) % 81)) && (((((int)threadIdx.x) + 68) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 5) % 9))) && (((((int)threadIdx.x) + 5) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 392) / 81) * 49)) + ((((((int)threadIdx.x) + 68) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 5) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 448)] = (((((9 <= ((((int)threadIdx.x) + 43) % 81)) && (((((int)threadIdx.x) + 43) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 7) % 9))) && (((((int)threadIdx.x) + 7) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 448) / 81) * 49)) + ((((((int)threadIdx.x) + 43) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 7) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 504)] = ((((((int)threadIdx.x) < 54) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 504) / 81) * 49)) + ((((int)threadIdx.x) / 9) * 7)) + (((int)threadIdx.x) % 9)) + 6)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 560)] = (((((9 <= ((((int)threadIdx.x) + 74) % 81)) && (((((int)threadIdx.x) + 74) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 2) % 9))) && (((((int)threadIdx.x) + 2) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 560) / 81) * 49)) + ((((((int)threadIdx.x) + 74) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 2) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 616)] = (((((9 <= ((((int)threadIdx.x) + 49) % 81)) && (((((int)threadIdx.x) + 49) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 4) % 9))) && (((((int)threadIdx.x) + 4) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 616) / 81) * 49)) + ((((((int)threadIdx.x) + 49) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 4) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 672)] = ((((((int)threadIdx.x) < 48) && (1 <= ((((int)threadIdx.x) + 6) % 9))) && (((((int)threadIdx.x) + 6) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 672) / 81) * 49)) + (((((int)threadIdx.x) + 24) / 9) * 7)) + ((((int)threadIdx.x) + 6) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 728)] = (((((9 <= ((((int)threadIdx.x) + 80) % 81)) && (((((int)threadIdx.x) + 80) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 8) % 9))) && (((((int)threadIdx.x) + 8) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 728) / 81) * 49)) + ((((((int)threadIdx.x) + 80) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 8) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 784)] = (((((9 <= ((((int)threadIdx.x) + 55) % 81)) && (((((int)threadIdx.x) + 55) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 1) % 9))) && (((((int)threadIdx.x) + 1) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 784) / 81) * 49)) + ((((((int)threadIdx.x) + 55) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 1) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 840)] = (((((9 <= ((((int)threadIdx.x) + 30) % 81)) && (((((int)threadIdx.x) + 30) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 3) % 9))) && (((((int)threadIdx.x) + 3) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 840) / 81) * 49)) + ((((((int)threadIdx.x) + 30) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 3) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 896)] = ((((4 <= ((int)threadIdx.x)) && (1 <= ((((int)threadIdx.x) + 5) % 9))) && (((((int)threadIdx.x) + 5) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 896) / 81) * 49)) + (((((int)threadIdx.x) + 5) / 9) * 7)) + ((((int)threadIdx.x) + 5) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 952)] = (((((9 <= ((((int)threadIdx.x) + 61) % 81)) && (((((int)threadIdx.x) + 61) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 7) % 9))) && (((((int)threadIdx.x) + 7) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 952) / 81) * 49)) + ((((((int)threadIdx.x) + 61) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 7) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1008)] = (((((1 <= (((((int)threadIdx.x) / 9) + 4) % 9)) && (((((int)threadIdx.x) + 36) % 81) < 72)) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1008) / 81) * 49)) + ((((((int)threadIdx.x) / 9) + 4) % 9) * 7)) + (((int)threadIdx.x) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1064)] = (((1 <= ((((int)threadIdx.x) + 2) % 9)) && (((((int)threadIdx.x) + 2) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1064) / 81) * 49)) + (((((int)threadIdx.x) + 11) / 9) * 7)) + ((((int)threadIdx.x) + 2) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1120)] = (((((9 <= ((((int)threadIdx.x) + 67) % 81)) && (((((int)threadIdx.x) + 67) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 4) % 9))) && (((((int)threadIdx.x) + 4) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1120) / 81) * 49)) + ((((((int)threadIdx.x) + 67) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 4) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1176)] = (((((9 <= ((((int)threadIdx.x) + 42) % 81)) && (((((int)threadIdx.x) + 42) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 6) % 9))) && (((((int)threadIdx.x) + 6) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1176) / 81) * 49)) + ((((((int)threadIdx.x) + 42) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 6) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1232)] = ((((((int)threadIdx.x) < 55) && (1 <= ((((int)threadIdx.x) + 8) % 9))) && (((((int)threadIdx.x) + 8) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1232) / 81) * 49)) + (((((int)threadIdx.x) + 17) / 9) * 7)) + ((((int)threadIdx.x) + 8) % 9)) - 8)] : 0.000000e+00f);
- if (((int)threadIdx.x) < 8) {
- pad_temp_shared[(((int)threadIdx.x) + 1288)] = 0.000000e+00f;
- }
- kernel_shared[((int)threadIdx.x)] = kernel[(((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x))];
- kernel_shared[(((int)threadIdx.x) + 56)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 112)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 112) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 168)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 168) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 224)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 224) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 280)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 280) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 336)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 336) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 392)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 392) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 448)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 448) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 504)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 504) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 560)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 560) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 616)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 616) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 672)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 672) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 728)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 728) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 784)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 784) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 840)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 840) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 896)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 896) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 952)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 952) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1008)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 32256)];
- kernel_shared[(((int)threadIdx.x) + 1064)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1064) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1120)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1120) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1176)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1176) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 1232)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1232) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1288)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1288) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1344)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1344) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 1400)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1400) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1456)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1456) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1512)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1512) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 1568)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1568) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1624)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1624) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1680)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1680) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1736)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1736) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1792)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1792) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1848)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1848) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1904)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1904) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1960)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1960) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2016)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 64512)];
- kernel_shared[(((int)threadIdx.x) + 2072)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2072) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2128)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2128) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2184)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2184) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 2240)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2240) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2296)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2296) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2352)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2352) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 2408)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2408) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2464)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2464) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2520)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2520) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 2576)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2576) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2632)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2632) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2688)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2688) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2744)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2744) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2800)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2800) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2856)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2856) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2912)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2912) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2968)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2968) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3024)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 96768)];
- kernel_shared[(((int)threadIdx.x) + 3080)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3080) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3136)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3136) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3192)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3192) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 3248)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3248) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3304)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3304) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3360)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3360) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 3416)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3416) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3472)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3472) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3528)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3528) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 3584)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3584) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3640)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3640) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3696)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3696) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3752)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3752) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3808)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3808) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3864)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3864) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3920)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3920) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3976)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3976) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4032)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 129024)];
- kernel_shared[(((int)threadIdx.x) + 4088)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4088) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4144)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4144) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4200)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4200) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 4256)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4256) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4312)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4312) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4368)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4368) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 4424)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4424) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4480)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4480) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4536)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4536) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- if (((int)threadIdx.x) < 16) {
- kernel_shared[(((int)threadIdx.x) + 4592)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4592) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 128) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- }
- __syncthreads();
- for (int rc_outer_inner = 0; rc_outer_inner < 4; ++rc_outer_inner) {
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- }
- }
- for (int i1_inner = 0; i1_inner < 2; ++i1_inner) {
- for (int i3_inner = 0; i3_inner < 7; ++i3_inner) {
- compute[(((((((int)blockIdx.x) * 1568) + ((((int)threadIdx.x) / 7) * 98)) + (i1_inner * 49)) + ((((int)threadIdx.x) % 7) * 7)) + i3_inner)] = max((conv2d_nchw[((i1_inner * 7) + i3_inner)] + bias[(((((int)blockIdx.x) * 32) + ((((int)threadIdx.x) / 7) * 2)) + i1_inner)]), 0.000000e+00f);
- compute[((((((((int)blockIdx.x) * 1568) + ((((int)threadIdx.x) / 7) * 98)) + (i1_inner * 49)) + ((((int)threadIdx.x) % 7) * 7)) + i3_inner) + 784)] = max((conv2d_nchw[(((i1_inner * 7) + i3_inner) + 14)] + bias[((((((int)blockIdx.x) * 32) + ((((int)threadIdx.x) / 7) * 2)) + i1_inner) + 16)]), 0.000000e+00f);
+ for (int ry_outer_outer = 0; ry_outer_outer < 3; ++ry_outer_outer) {
+ __syncthreads();
+ if (((int)threadIdx.x) < 144) {
+ pad_temp_shared[((int)threadIdx.x)] = (((((1 <= (ry_outer_outer + (((int)blockIdx.x) % 7))) && ((ry_outer_outer + (((int)blockIdx.x) % 7)) < 8)) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[((((((rc_outer_outer * 784) + ((((int)threadIdx.x) / 9) * 49)) + (ry_outer_outer * 7)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 9)) - 8)] : 0.000000e+00f);
+ }
+ kernel_shared[((int)threadIdx.x)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 224)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 224) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 448)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 448) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 672)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 64512)];
+ kernel_shared[(((int)threadIdx.x) + 896)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 896) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 1120)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 1120) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 1344)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 129024)];
+ kernel_shared[(((int)threadIdx.x) + 1568)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 1568) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 1792)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 1792) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 2016)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 193536)];
+ kernel_shared[(((int)threadIdx.x) + 2240)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 2240) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 2464)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 2464) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 2688)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 258048)];
+ kernel_shared[(((int)threadIdx.x) + 2912)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 2912) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 3136)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 3136) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 3360)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 322560)];
+ kernel_shared[(((int)threadIdx.x) + 3584)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 3584) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 3808)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 3808) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 4032)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 387072)];
+ kernel_shared[(((int)threadIdx.x) + 4256)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 4256) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 4480)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 4480) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 4704)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 451584)];
+ kernel_shared[(((int)threadIdx.x) + 4928)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 4928) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 5152)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 5152) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 5376)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 516096)];
+ kernel_shared[(((int)threadIdx.x) + 5600)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 5600) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 5824)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 5824) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ if (((int)threadIdx.x) < 96) {
+ kernel_shared[(((int)threadIdx.x) + 6048)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 580608)];
+ }
+ __syncthreads();
+ for (int rc_outer_inner = 0; rc_outer_inner < 2; ++rc_outer_inner) {
+ for (int rx_outer_inner = 0; rx_outer_inner < 3; ++rx_outer_inner) {
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1536)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3072)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4608)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1539)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3075)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4611)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 6)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1542)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3078)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4614)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 9)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1545)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3081)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4617)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 12)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1548)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3084)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4620)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 15)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1551)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3087)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4623)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 18)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1554)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3090)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4626)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 21)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1557)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3093)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4629)]));
+ }
+ }
}
}
+ compute[(((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7))] = max((conv2d_nchw[0] + bias[(((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7))]), 0.000000e+00f);
+ compute[((((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7)) + 1568)] = max((conv2d_nchw[1] + bias[((((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7)) + 32)]), 0.000000e+00f);
+ compute[((((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7)) + 3136)] = max((conv2d_nchw[2] + bias[((((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7)) + 64)]), 0.000000e+00f);
+ compute[((((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7)) + 4704)] = max((conv2d_nchw[3] + bias[((((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7)) + 96)]), 0.000000e+00f);
}
@@ -2913,7 +680,7 @@ In the example below we resume the status and do more 5 trials.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 5 minutes 42.541 seconds)
+ **Total running time of the script:** ( 5 minutes 27.158 seconds)
.. _sphx_glr_download_how_to_tune_with_autoscheduler_tune_conv2d_layer_cuda.py:
diff --git a/docs/_sources/how_to/tune_with_autoscheduler/tune_network_cuda.rst.txt b/docs/_sources/how_to/tune_with_autoscheduler/tune_network_cuda.rst.txt
index 13caa8442d..e3e964648d 100644
--- a/docs/_sources/how_to/tune_with_autoscheduler/tune_network_cuda.rst.txt
+++ b/docs/_sources/how_to/tune_with_autoscheduler/tune_network_cuda.rst.txt
@@ -643,7 +643,7 @@ so we can read the log file and load the best schedules.
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 7.8418 7.8454 7.8478 7.8321 0.0069
+ 7.9013 7.9087 7.9108 7.8845 0.0119
@@ -671,7 +671,7 @@ Other Tips
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 1.296 seconds)
+ **Total running time of the script:** ( 1 minutes 1.762 seconds)
.. _sphx_glr_download_how_to_tune_with_autoscheduler_tune_network_cuda.py:
diff --git a/docs/_sources/how_to/tune_with_autoscheduler/tune_network_x86.rst.txt b/docs/_sources/how_to/tune_with_autoscheduler/tune_network_x86.rst.txt
index 3555362ab2..e02576b14d 100644
--- a/docs/_sources/how_to/tune_with_autoscheduler/tune_network_x86.rst.txt
+++ b/docs/_sources/how_to/tune_with_autoscheduler/tune_network_x86.rst.txt
@@ -662,7 +662,7 @@ so we can read the log file and load the best schedules.
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 766.8676 768.5019 769.7074 762.3936 3.2017
+ 751.4864 753.3622 753.5823 747.5147 2.8099
@@ -690,7 +690,7 @@ Other Tips
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 31.335 seconds)
+ **Total running time of the script:** ( 1 minutes 31.258 seconds)
.. _sphx_glr_download_how_to_tune_with_autoscheduler_tune_network_x86.py:
diff --git a/docs/_sources/how_to/tune_with_autoscheduler/tune_sparse_x86.rst.txt b/docs/_sources/how_to/tune_with_autoscheduler/tune_sparse_x86.rst.txt
index be903a9be3..3957bd791d 100644
--- a/docs/_sources/how_to/tune_with_autoscheduler/tune_sparse_x86.rst.txt
+++ b/docs/_sources/how_to/tune_with_autoscheduler/tune_sparse_x86.rst.txt
@@ -386,337 +386,27 @@ layout transformation, parallelization, vectorization, unrolling, and operator f
placeholder_4: Buffer(placeholder_14: Pointer(float32), float32, [128, 512], []),
compute: Buffer(compute_2: Pointer(float32), float32, [128, 512], [])}
buffer_map = {placeholder_5: placeholder, placeholder_6: placeholder_1, placeholder_7: placeholder_2, placeholder_8: placeholder_3, placeholder_9: placeholder_4, compute_1: compute} {
- for (i0.outer.i1.outer.fused: int32, 0, 32) "parallel" {
- allocate(compute_3: Pointer(global float32), float32, [2048]), storage_scope = global {
- for (i.outer.inner: int32, 0, 32) {
- let cse_var_1: int32 = (i.outer.inner*64)
- {
- compute_4: Buffer(compute_3, float32, [2048], [])[cse_var_1] = 0f32
- compute_4[(cse_var_1 + 1)] = 0f32
- compute_4[(cse_var_1 + 2)] = 0f32
- compute_4[(cse_var_1 + 3)] = 0f32
- compute_4[(cse_var_1 + 4)] = 0f32
- compute_4[(cse_var_1 + 5)] = 0f32
- compute_4[(cse_var_1 + 6)] = 0f32
- compute_4[(cse_var_1 + 7)] = 0f32
- compute_4[(cse_var_1 + 8)] = 0f32
- compute_4[(cse_var_1 + 9)] = 0f32
- compute_4[(cse_var_1 + 10)] = 0f32
- compute_4[(cse_var_1 + 11)] = 0f32
- compute_4[(cse_var_1 + 12)] = 0f32
- compute_4[(cse_var_1 + 13)] = 0f32
- compute_4[(cse_var_1 + 14)] = 0f32
- compute_4[(cse_var_1 + 15)] = 0f32
- compute_4[(cse_var_1 + 16)] = 0f32
- compute_4[(cse_var_1 + 17)] = 0f32
- compute_4[(cse_var_1 + 18)] = 0f32
- compute_4[(cse_var_1 + 19)] = 0f32
- compute_4[(cse_var_1 + 20)] = 0f32
- compute_4[(cse_var_1 + 21)] = 0f32
- compute_4[(cse_var_1 + 22)] = 0f32
- compute_4[(cse_var_1 + 23)] = 0f32
- compute_4[(cse_var_1 + 24)] = 0f32
- compute_4[(cse_var_1 + 25)] = 0f32
- compute_4[(cse_var_1 + 26)] = 0f32
- compute_4[(cse_var_1 + 27)] = 0f32
- compute_4[(cse_var_1 + 28)] = 0f32
- compute_4[(cse_var_1 + 29)] = 0f32
- compute_4[(cse_var_1 + 30)] = 0f32
- compute_4[(cse_var_1 + 31)] = 0f32
- compute_4[(cse_var_1 + 32)] = 0f32
- compute_4[(cse_var_1 + 33)] = 0f32
- compute_4[(cse_var_1 + 34)] = 0f32
- compute_4[(cse_var_1 + 35)] = 0f32
- compute_4[(cse_var_1 + 36)] = 0f32
- compute_4[(cse_var_1 + 37)] = 0f32
- compute_4[(cse_var_1 + 38)] = 0f32
- compute_4[(cse_var_1 + 39)] = 0f32
- compute_4[(cse_var_1 + 40)] = 0f32
- compute_4[(cse_var_1 + 41)] = 0f32
- compute_4[(cse_var_1 + 42)] = 0f32
- compute_4[(cse_var_1 + 43)] = 0f32
- compute_4[(cse_var_1 + 44)] = 0f32
- compute_4[(cse_var_1 + 45)] = 0f32
- compute_4[(cse_var_1 + 46)] = 0f32
- compute_4[(cse_var_1 + 47)] = 0f32
- compute_4[(cse_var_1 + 48)] = 0f32
- compute_4[(cse_var_1 + 49)] = 0f32
- compute_4[(cse_var_1 + 50)] = 0f32
- compute_4[(cse_var_1 + 51)] = 0f32
- compute_4[(cse_var_1 + 52)] = 0f32
- compute_4[(cse_var_1 + 53)] = 0f32
- compute_4[(cse_var_1 + 54)] = 0f32
- compute_4[(cse_var_1 + 55)] = 0f32
- compute_4[(cse_var_1 + 56)] = 0f32
- compute_4[(cse_var_1 + 57)] = 0f32
- compute_4[(cse_var_1 + 58)] = 0f32
- compute_4[(cse_var_1 + 59)] = 0f32
- compute_4[(cse_var_1 + 60)] = 0f32
- compute_4[(cse_var_1 + 61)] = 0f32
- compute_4[(cse_var_1 + 62)] = 0f32
- compute_4[(cse_var_1 + 63)] = 0f32
- for (elem_idx: int32, 0, (placeholder_15: Buffer(placeholder_13, int32, [33], [])[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])) {
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- compute_4[cse_var_1] = (compute_4[cse_var_1] + (placeholder_16: Buffer(placeholder_11, float32, [78656], [])[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17: Buffer(placeholder_10, float32, [32768], [])[((i.outer.inner*1024) + placeholder_18: Buffer(placeholder_12, int32, [4916], [])[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_2: int32 = (cse_var_1 + 1)
- compute_4[cse_var_2] = (compute_4[cse_var_2] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_3: int32 = (cse_var_1 + 2)
- compute_4[cse_var_3] = (compute_4[cse_var_3] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_4: int32 = (cse_var_1 + 3)
- compute_4[cse_var_4] = (compute_4[cse_var_4] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_5: int32 = (cse_var_1 + 4)
- compute_4[cse_var_5] = (compute_4[cse_var_5] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_6: int32 = (cse_var_1 + 5)
- compute_4[cse_var_6] = (compute_4[cse_var_6] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_7: int32 = (cse_var_1 + 6)
- compute_4[cse_var_7] = (compute_4[cse_var_7] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_8: int32 = (cse_var_1 + 7)
- compute_4[cse_var_8] = (compute_4[cse_var_8] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_9: int32 = (cse_var_1 + 8)
- compute_4[cse_var_9] = (compute_4[cse_var_9] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_10: int32 = (cse_var_1 + 9)
- compute_4[cse_var_10] = (compute_4[cse_var_10] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_11: int32 = (cse_var_1 + 10)
- compute_4[cse_var_11] = (compute_4[cse_var_11] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_12: int32 = (cse_var_1 + 11)
- compute_4[cse_var_12] = (compute_4[cse_var_12] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_13: int32 = (cse_var_1 + 12)
- compute_4[cse_var_13] = (compute_4[cse_var_13] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_14: int32 = (cse_var_1 + 13)
- compute_4[cse_var_14] = (compute_4[cse_var_14] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_15: int32 = (cse_var_1 + 14)
- compute_4[cse_var_15] = (compute_4[cse_var_15] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_16: int32 = (cse_var_1 + 15)
- compute_4[cse_var_16] = (compute_4[cse_var_16] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_17: int32 = (cse_var_1 + 16)
- compute_4[cse_var_17] = (compute_4[cse_var_17] + (placeholder_16[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_18: int32 = (cse_var_1 + 17)
- compute_4[cse_var_18] = (compute_4[cse_var_18] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_19: int32 = (cse_var_1 + 18)
- compute_4[cse_var_19] = (compute_4[cse_var_19] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_20: int32 = (cse_var_1 + 19)
- compute_4[cse_var_20] = (compute_4[cse_var_20] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_21: int32 = (cse_var_1 + 20)
- compute_4[cse_var_21] = (compute_4[cse_var_21] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_22: int32 = (cse_var_1 + 21)
- compute_4[cse_var_22] = (compute_4[cse_var_22] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_23: int32 = (cse_var_1 + 22)
- compute_4[cse_var_23] = (compute_4[cse_var_23] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_24: int32 = (cse_var_1 + 23)
- compute_4[cse_var_24] = (compute_4[cse_var_24] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_25: int32 = (cse_var_1 + 24)
- compute_4[cse_var_25] = (compute_4[cse_var_25] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_26: int32 = (cse_var_1 + 25)
- compute_4[cse_var_26] = (compute_4[cse_var_26] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_27: int32 = (cse_var_1 + 26)
- compute_4[cse_var_27] = (compute_4[cse_var_27] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_28: int32 = (cse_var_1 + 27)
- compute_4[cse_var_28] = (compute_4[cse_var_28] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_29: int32 = (cse_var_1 + 28)
- compute_4[cse_var_29] = (compute_4[cse_var_29] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_30: int32 = (cse_var_1 + 29)
- compute_4[cse_var_30] = (compute_4[cse_var_30] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_31: int32 = (cse_var_1 + 30)
- compute_4[cse_var_31] = (compute_4[cse_var_31] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_32: int32 = (cse_var_1 + 31)
- compute_4[cse_var_32] = (compute_4[cse_var_32] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_33: int32 = (cse_var_1 + 32)
- compute_4[cse_var_33] = (compute_4[cse_var_33] + (placeholder_16[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_34: int32 = (cse_var_1 + 33)
- compute_4[cse_var_34] = (compute_4[cse_var_34] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_35: int32 = (cse_var_1 + 34)
- compute_4[cse_var_35] = (compute_4[cse_var_35] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_36: int32 = (cse_var_1 + 35)
- compute_4[cse_var_36] = (compute_4[cse_var_36] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_37: int32 = (cse_var_1 + 36)
- compute_4[cse_var_37] = (compute_4[cse_var_37] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_38: int32 = (cse_var_1 + 37)
- compute_4[cse_var_38] = (compute_4[cse_var_38] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_39: int32 = (cse_var_1 + 38)
- compute_4[cse_var_39] = (compute_4[cse_var_39] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_40: int32 = (cse_var_1 + 39)
- compute_4[cse_var_40] = (compute_4[cse_var_40] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_41: int32 = (cse_var_1 + 40)
- compute_4[cse_var_41] = (compute_4[cse_var_41] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_42: int32 = (cse_var_1 + 41)
- compute_4[cse_var_42] = (compute_4[cse_var_42] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_43: int32 = (cse_var_1 + 42)
- compute_4[cse_var_43] = (compute_4[cse_var_43] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_44: int32 = (cse_var_1 + 43)
- compute_4[cse_var_44] = (compute_4[cse_var_44] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_45: int32 = (cse_var_1 + 44)
- compute_4[cse_var_45] = (compute_4[cse_var_45] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_46: int32 = (cse_var_1 + 45)
- compute_4[cse_var_46] = (compute_4[cse_var_46] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_47: int32 = (cse_var_1 + 46)
- compute_4[cse_var_47] = (compute_4[cse_var_47] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_48: int32 = (cse_var_1 + 47)
- compute_4[cse_var_48] = (compute_4[cse_var_48] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_49: int32 = (cse_var_1 + 48)
- compute_4[cse_var_49] = (compute_4[cse_var_49] + (placeholder_16[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_50: int32 = (cse_var_1 + 49)
- compute_4[cse_var_50] = (compute_4[cse_var_50] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_51: int32 = (cse_var_1 + 50)
- compute_4[cse_var_51] = (compute_4[cse_var_51] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_52: int32 = (cse_var_1 + 51)
- compute_4[cse_var_52] = (compute_4[cse_var_52] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_53: int32 = (cse_var_1 + 52)
- compute_4[cse_var_53] = (compute_4[cse_var_53] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_54: int32 = (cse_var_1 + 53)
- compute_4[cse_var_54] = (compute_4[cse_var_54] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_55: int32 = (cse_var_1 + 54)
- compute_4[cse_var_55] = (compute_4[cse_var_55] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_56: int32 = (cse_var_1 + 55)
- compute_4[cse_var_56] = (compute_4[cse_var_56] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_57: int32 = (cse_var_1 + 56)
- compute_4[cse_var_57] = (compute_4[cse_var_57] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_58: int32 = (cse_var_1 + 57)
- compute_4[cse_var_58] = (compute_4[cse_var_58] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_59: int32 = (cse_var_1 + 58)
- compute_4[cse_var_59] = (compute_4[cse_var_59] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_60: int32 = (cse_var_1 + 59)
- compute_4[cse_var_60] = (compute_4[cse_var_60] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_61: int32 = (cse_var_1 + 60)
- compute_4[cse_var_61] = (compute_4[cse_var_61] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_62: int32 = (cse_var_1 + 61)
- compute_4[cse_var_62] = (compute_4[cse_var_62] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_63: int32 = (cse_var_1 + 62)
- compute_4[cse_var_63] = (compute_4[cse_var_63] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_64: int32 = (cse_var_1 + 63)
- compute_4[cse_var_64] = (compute_4[cse_var_64] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
+ for (i0.outer.i1.outer.fused: int32, 0, 1024) "parallel" {
+ allocate(compute_3: Pointer(global float32), float32, [64]), storage_scope = global {
+ for (i.inner.init: int32, 0, 4) {
+ for (j.init: int32, 0, 16) {
+ compute_4: Buffer(compute_3, float32, [64], [])[((i.inner.init*16) + j.init)] = 0f32
+ }
+ }
+ for (elem_idx: int32, 0, let cse_var_1: int32 = floormod(i0.outer.i1.outer.fused, 32) in (placeholder_15: Buffer(placeholder_13, int32, [33], [])[(cse_var_1 + 1)] - placeholder_15[cse_var_1])) {
+ for (i.inner: int32, 0, 4) {
+ for (j: int32, 0, 16) {
+ let cse_var_2: int32 = floormod(i0.outer.i1.outer.fused, 32)
+ if @tir.likely((elem_idx < (placeholder_15[(cse_var_2 + 1)] - placeholder_15[cse_var_2])), dtype=bool) {
+ let cse_var_3: int32 = ((i.inner*16) + j)
+ compute_4[cse_var_3] = (compute_4[cse_var_3] + (placeholder_16: Buffer(placeholder_11, float32, [78656], [])[(((placeholder_15[cse_var_2]*16) + (elem_idx*16)) + j)]*max(placeholder_17: Buffer(placeholder_10, float32, [32768], [])[(((floordiv(i0.outer.i1.outer.fused, 32)*1024) + (i.inner*256)) + placeholder_18: Buffer(placeholder_12, int32, [4916], [])[(placeholder_15[cse_var_2] + elem_idx)])], 0f32)))
}
}
}
}
- for (i0.inner: int32, 0, 128) {
- let cse_var_65: int32 = ((i0.inner*512) + (i0.outer.i1.outer.fused*16))
- compute_5: Buffer(compute_2, float32, [65536], [])[ramp(cse_var_65, 1, 16)] = max((compute_4[ramp((i0.inner*16), 1, 16)] + placeholder_19: Buffer(placeholder_14, float32, [65536], [])[ramp(cse_var_65, 1, 16)]), broadcast(0f32, 16))
+ for (i0.inner: int32, 0, 4) {
+ let cse_var_4: int32 = (((floordiv(i0.outer.i1.outer.fused, 32)*2048) + (i0.inner*512)) + (floormod(i0.outer.i1.outer.fused, 32)*16))
+ compute_5: Buffer(compute_2, float32, [65536], [])[ramp(cse_var_4, 1, 16)] = max((compute_4[ramp((i0.inner*16), 1, 16)] + placeholder_19: Buffer(placeholder_14, float32, [65536], [])[ramp(cse_var_4, 1, 16)]), broadcast(0f32, 16))
}
}
}
@@ -772,7 +462,7 @@ We build the binary and check its correctness and performance.
.. code-block:: none
- Execution time of this operator: 3.211 ms
+ Execution time of this operator: 1.333 ms
diff --git a/docs/_sources/how_to/tune_with_autotvm/sg_execution_times.rst.txt b/docs/_sources/how_to/tune_with_autotvm/sg_execution_times.rst.txt
index 4af2002bb3..6941a406b7 100644
--- a/docs/_sources/how_to/tune_with_autotvm/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/tune_with_autotvm/sg_execution_times.rst.txt
@@ -5,16 +5,16 @@
Computation times
=================
-**00:26.988** total execution time for **how_to_tune_with_autotvm** files:
+**00:26.542** total execution time for **how_to_tune_with_autotvm** files:
+--------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_conv2d_cuda.py` (``tune_conv2d_cuda.py``) | 00:26.949 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_conv2d_cuda.py` (``tune_conv2d_cuda.py``) | 00:26.507 | 0.0 MB |
+--------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_relay_x86.py` (``tune_relay_x86.py``) | 00:00.024 | 0.0 MB |
+| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_relay_x86.py` (``tune_relay_x86.py``) | 00:00.020 | 0.0 MB |
+--------------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_relay_cuda.py` (``tune_relay_cuda.py``) | 00:00.005 | 0.0 MB |
+--------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_relay_arm.py` (``tune_relay_arm.py``) | 00:00.005 | 0.0 MB |
-+--------------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_relay_mobile_gpu.py` (``tune_relay_mobile_gpu.py``) | 00:00.005 | 0.0 MB |
+--------------------------------------------------------------------------------------------------+-----------+--------+
+| :ref:`sphx_glr_how_to_tune_with_autotvm_tune_relay_arm.py` (``tune_relay_arm.py``) | 00:00.005 | 0.0 MB |
++--------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/tune_with_autotvm/tune_conv2d_cuda.rst.txt b/docs/_sources/how_to/tune_with_autotvm/tune_conv2d_cuda.rst.txt
index 382a0e0c42..fe432021ac 100644
--- a/docs/_sources/how_to/tune_with_autotvm/tune_conv2d_cuda.rst.txt
+++ b/docs/_sources/how_to/tune_with_autotvm/tune_conv2d_cuda.rst.txt
@@ -387,9 +387,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 1, 4]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 128, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1255863
- No: 2 GFLOPS: 103.32/103.32 result: MeasureResult(costs=(0.0022406029555555556,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.7429840564727783, timestamp=1670933521.5859842) [('tile_f', [-1, 1, 16, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,77914
- No: 3 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 4, 128, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,5600811
+ No: 2 GFLOPS: 0.00/0.00 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -511,8 +510,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 8, 8]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 4, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,5851937
- No: 4 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 32, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 1, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,4877441
+ No: 3 GFLOPS: 0.00/0.00 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -634,8 +633,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 4, 2]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 8]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,5140155
- No: 5 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 64, 2]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6378114
+ No: 4 GFLOPS: 0.00/0.00 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -757,8 +756,10 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 64, 2, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 16, 32]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1512956
- No: 6 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 64, 1, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 2, 32]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1500626
+ No: 5 GFLOPS: 107.57/107.57 result: MeasureResult(costs=(0.002152012914893617,), error_no=MeasureErrorNo.NO_ERROR, all_cost=3.096332550048828, timestamp=1670949979.592225) [('tile_f', [-1, 2, 8, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9371368
+ No: 6 GFLOPS: 47.50/107.57 result: MeasureResult(costs=(0.004874122619047619,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.0825843811035156, timestamp=1670949981.2732048) [('tile_f', [-1, 1, 32, 16]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 2, 1]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,586264
+ No: 7 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -880,8 +881,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 16, 1, 8]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 32, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,10122560
- No: 7 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 4, 32]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 16, 16]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,4975054
+ No: 8 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1003,8 +1004,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 64, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6956666
- No: 8 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 2, 4]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 1, 256]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2120688
+ No: 9 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1126,9 +1127,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 2, 128]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 128, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6064734
- No: 9 GFLOPS: 94.26/103.32 result: MeasureResult(costs=(0.0024560546829268293,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.2918369770050049, timestamp=1670933525.4022467) [('tile_f', [-1, 1, 64, 1]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 2, 32]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,146125
- No: 10 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 2, 16]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 4, 64]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,3459450
+ No: 10 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1250,8 +1250,9 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 8, 2]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 16, 4]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,3955902
- No: 11 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 4, 32]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 8, 1]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,3304155
+ No: 11 GFLOPS: 32.94/107.57 result: MeasureResult(costs=(0.007028339066666666,), error_no=MeasureErrorNo.NO_ERROR, all_cost=2.3546528816223145, timestamp=1670949984.7977166) [('tile_f', [-1, 1, 1, 2]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 32, 2]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,7992435
+ No: 12 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1373,11 +1374,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 2, 32]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 2, 32]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9438633
- No: 12 GFLOPS: 27.72/103.32 result: MeasureResult(costs=(0.008351099105263158,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.592003583908081, timestamp=1670933526.43331) [('tile_f', [-1, 1, 1, 8]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 4, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1397576
- No: 13 GFLOPS: 9.08/103.32 result: MeasureResult(costs=(0.025502439249999998,), error_no=MeasureErrorNo.NO_ERROR, all_cost=2.2341725826263428, timestamp=1670933528.8338175) [('tile_f', [-1, 4, 8, 2]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 1]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2141781
- No: 14 GFLOPS: 132.18/132.18 result: MeasureResult(costs=(0.0017514406195652176,), error_no=MeasureErrorNo.NO_ERROR, all_cost=2.2057106494903564, timestamp=1670933529.8254948) [('tile_f', [-1, 16, 4, 2]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 16, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,3535916
- No: 15 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 128, 4, 1]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 128, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,643086
+ No: 13 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1499,8 +1497,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 64, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 32, 1]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9698968
- No: 16 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 1, 128]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 32, 8]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,8633011
+ No: 14 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1622,8 +1620,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 4, 32, 2]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 8, 1]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2529432
- No: 17 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 32, 4, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 2, 256]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1351044
+ No: 15 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1745,8 +1743,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 32, 4]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 8, 32]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,7316451
- No: 18 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 128, 1, 4]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 1, 512]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6774567
+ No: 16 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1868,8 +1866,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 16, 2, 1]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1341794
- No: 19 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 128, 2]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 256, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2387978
+ No: 17 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1991,8 +1989,8 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 32, 16, 1]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 16, 1]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,8338919
- No: 20 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 4, 1, 16]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 8, 32]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,10025566
+ No: 18 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -2114,7 +2112,253 @@ for this template
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
- tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 4, 8]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 1, 256]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,957590
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 512, 1]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6081734
+ No: 19 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
+ func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
+ func = build(s, args, target_host=task.target_host, runtime=runtime)
+ File "/workspace/python/tvm/driver/build_module.py", line 227, in build
+ input_mod = lower(inputs, args, name=name, binds=binds)
+ File "/workspace/python/tvm/driver/build_module.py", line 134, in lower
+ return ffi.lower_schedule(inp, args, name, binds, simple_mode)
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 331, in tvm._ffi._cy3.core.PackedFuncBase.__call__
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 276, in tvm._ffi._cy3.core.FuncCall
+ File "tvm/_ffi/_cython/./base.pxi", line 181, in tvm._ffi._cy3.core.CHECK_CALL
+ tvm._ffi.base.TVMError: Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel
+
+ Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 512, 1, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 256, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1419669
+ No: 20 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
+ func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
+ func = build(s, args, target_host=task.target_host, runtime=runtime)
+ File "/workspace/python/tvm/driver/build_module.py", line 227, in build
+ input_mod = lower(inputs, args, name=name, binds=binds)
+ File "/workspace/python/tvm/driver/build_module.py", line 134, in lower
+ return ffi.lower_schedule(inp, args, name, binds, simple_mode)
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 331, in tvm._ffi._cy3.core.PackedFuncBase.__call__
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 276, in tvm._ffi._cy3.core.FuncCall
+ File "tvm/_ffi/_cython/./base.pxi", line 181, in tvm._ffi._cy3.core.CHECK_CALL
+ tvm._ffi.base.TVMError: Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel
+
+ Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+ tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 16, 8]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 32, 16]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9043478
@@ -2169,9 +2413,9 @@ and measure running time.
Finish loading 20 records
Best config:
- [('tile_f', [-1, 16, 4, 2]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 16, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,3535916
+ [('tile_f', [-1, 2, 8, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9371368
Finish loading 20 records
- Time cost of this operator: 0.002177
+ Time cost of this operator: 0.001173
diff --git a/docs/_sources/how_to/work_with_microtvm/micro_autotune.rst.txt b/docs/_sources/how_to/work_with_microtvm/micro_autotune.rst.txt
index 41c8ff581e..1595306f29 100644
--- a/docs/_sources/how_to/work_with_microtvm/micro_autotune.rst.txt
+++ b/docs/_sources/how_to/work_with_microtvm/micro_autotune.rst.txt
@@ -329,10 +329,10 @@ Timing the untuned program
########## Build without Autotuning ##########
Node Name Ops Time(us) Time(%) Shape Inputs Outputs Measurements(us)
--------- --- -------- ------- ----- ------ ------- ----------------
- tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 310.3 98.636 (1, 2, 10, 10, 3) 2 1 [310.3]
- tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 3.164 1.006 (1, 6, 10, 10) 1 1 [3.164]
- tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 1.127 0.358 (1, 1, 10, 10, 3) 1 1 [1.127]
- Total_time - 314.592 - - - - -
+ tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 311.0 98.728 (1, 2, 10, 10, 3) 2 1 [311.0]
+ tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 3.036 0.964 (1, 6, 10, 10) 1 1 [3.036]
+ tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 0.97 0.308 (1, 1, 10, 10, 3) 1 1 [0.97]
+ Total_time - 315.006 - - - - -
@@ -397,10 +397,10 @@ Timing the tuned program
########## Build with Autotuning ##########
Node Name Ops Time(us) Time(%) Shape Inputs Outputs Measurements(us)
--------- --- -------- ------- ----- ------ ------- ----------------
- tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 105.0 97.55 (1, 6, 10, 10, 1) 2 1 [105.0]
- tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 1.792 1.665 (1, 6, 10, 10) 1 1 [1.792]
- tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 0.845 0.785 (1, 3, 10, 10, 1) 1 1 [0.845]
- Total_time - 107.637 - - - - -
+ tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 100.1 97.276 (1, 6, 10, 10, 1) 2 1 [100.1]
+ tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 1.78 1.729 (1, 6, 10, 10) 1 1 [1.78]
+ tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 1.023 0.994 (1, 1, 10, 10, 3) 1 1 [1.023]
+ Total_time - 102.903 - - - - -
diff --git a/docs/_sources/how_to/work_with_microtvm/micro_pytorch.rst.txt b/docs/_sources/how_to/work_with_microtvm/micro_pytorch.rst.txt
index a7de9acedc..b2cd5b39af 100644
--- a/docs/_sources/how_to/work_with_microtvm/micro_pytorch.rst.txt
+++ b/docs/_sources/how_to/work_with_microtvm/micro_pytorch.rst.txt
@@ -109,7 +109,7 @@ download a cat image and preprocess it to use as the model input.
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torch/ao/quantization/utils.py:281: UserWarning: must run observer before calling calculate_qparams. Returning default values.
"must run observer before calling calculate_qparams. " +
Downloading: "https://download.pytorch.org/models/quantized/mobilenet_v2_qnnpack_37f702c5.pth" to /workspace/.cache/torch/hub/checkpoints/mobilenet_v2_qnnpack_37f702c5.pth
-
0%| | 0.00/3.42M [00:00<?, ?B/s]
100%|##########| 3.42M/3.42M [00:00<00:00, 61.6MB/s]
+
0%| | 0.00/3.42M [00:00<?, ?B/s]
70%|######9 | 2.39M/3.42M [00:00<00:00, 25.1MB/s]
100%|##########| 3.42M/3.42M [00:00<00:00, 34.3MB/s]
/workspace/python/tvm/relay/frontend/pytorch_utils.py:47: DeprecationWarning: distutils Version classes are deprecated. Use packaging.version instead.
return LooseVersion(torch_ver) > ver
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/setuptools/_distutils/version.py:346: DeprecationWarning: distutils Version classes are deprecated. Use packaging.version instead.
@@ -314,7 +314,7 @@ Look up prediction top 1 index in 1000 class synset.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 3.019 seconds)
+ **Total running time of the script:** ( 1 minutes 2.998 seconds)
.. _sphx_glr_download_how_to_work_with_microtvm_micro_pytorch.py:
diff --git a/docs/_sources/how_to/work_with_microtvm/micro_train.rst.txt b/docs/_sources/how_to/work_with_microtvm/micro_train.rst.txt
index 9761463d09..14eece5e21 100644
--- a/docs/_sources/how_to/work_with_microtvm/micro_train.rst.txt
+++ b/docs/_sources/how_to/work_with_microtvm/micro_train.rst.txt
@@ -225,7 +225,7 @@ take about **2 minutes** to download the Stanford Cars, while COCO 2017 validati
.. code-block:: none
- '/tmp/tmp_8kq7sx6/images/random'
+ '/tmp/tmpn64xoprz/images/random'
@@ -316,7 +316,7 @@ objects to other stuff? We can display some examples from our datasets using ``m
.. image-sg:: /how_to/work_with_microtvm/images/sphx_glr_micro_train_001.png
- :alt: [0.0, 1.0], [0.0, 1.0], [1.0, 0.0], [1.0, 0.0], [1.0, 0.0], [1.0, 0.0], [0.0, 1.0], [0.0, 1.0], [0.0, 1.0], [1.0, 0.0]
+ :alt: [1.0, 0.0], [0.0, 1.0], [0.0, 1.0], [1.0, 0.0], [1.0, 0.0], [0.0, 1.0], [1.0, 0.0], [1.0, 0.0], [1.0, 0.0], [0.0, 1.0]
:srcset: /how_to/work_with_microtvm/images/sphx_glr_micro_train_001.png
:class: sphx-glr-single-img
@@ -325,8 +325,8 @@ objects to other stuff? We can display some examples from our datasets using ``m
.. code-block:: none
- /tmp/tmp_8kq7sx6/images/target contains 8144 images
- /tmp/tmp_8kq7sx6/images/random contains 5000 images
+ /tmp/tmpn64xoprz/images/target contains 8144 images
+ /tmp/tmpn64xoprz/images/random contains 5000 images
@@ -501,13 +501,13 @@ the time on our validation set).
.. code-block:: none
Epoch 1/3
- 328/328 - 47s - loss: 0.2257 - accuracy: 0.9235 - val_loss: 0.2299 - val_accuracy: 0.9313 - 47s/epoch - 142ms/step
+ 328/328 - 47s - loss: 0.2277 - accuracy: 0.9188 - val_loss: 0.1379 - val_accuracy: 0.9535 - 47s/epoch - 144ms/step
Epoch 2/3
- 328/328 - 43s - loss: 0.0990 - accuracy: 0.9648 - val_loss: 0.1212 - val_accuracy: 0.9619 - 43s/epoch - 132ms/step
+ 328/328 - 43s - loss: 0.0917 - accuracy: 0.9676 - val_loss: 0.1415 - val_accuracy: 0.9573 - 43s/epoch - 132ms/step
Epoch 3/3
- 328/328 - 43s - loss: 0.0623 - accuracy: 0.9754 - val_loss: 0.1194 - val_accuracy: 0.9698 - 43s/epoch - 131ms/step
+ 328/328 - 43s - loss: 0.0686 - accuracy: 0.9754 - val_loss: 0.1399 - val_accuracy: 0.9607 - 43s/epoch - 132ms/step
- <keras.callbacks.History object at 0x7ff68c133210>
+ <keras.callbacks.History object at 0x7f8616e6cb50>
@@ -864,7 +864,7 @@ Arduino tutorial for how to do that `on GitHub <https://github.com/guberti/tvm-a
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 5 minutes 21.287 seconds)
+ **Total running time of the script:** ( 4 minutes 34.133 seconds)
.. _sphx_glr_download_how_to_work_with_microtvm_micro_train.py:
diff --git a/docs/_sources/how_to/work_with_microtvm/sg_execution_times.rst.txt b/docs/_sources/how_to/work_with_microtvm/sg_execution_times.rst.txt
index 84101d045a..83403fa469 100644
--- a/docs/_sources/how_to/work_with_microtvm/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/work_with_microtvm/sg_execution_times.rst.txt
@@ -5,18 +5,18 @@
Computation times
=================
-**07:26.613** total execution time for **how_to_work_with_microtvm** files:
+**06:40.496** total execution time for **how_to_work_with_microtvm** files:
+---------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_microtvm_micro_train.py` (``micro_train.py``) | 05:21.287 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_microtvm_micro_train.py` (``micro_train.py``) | 04:34.133 | 0.0 MB |
+---------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_microtvm_micro_pytorch.py` (``micro_pytorch.py``) | 01:03.019 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_microtvm_micro_pytorch.py` (``micro_pytorch.py``) | 01:02.998 | 0.0 MB |
+---------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_microtvm_micro_autotune.py` (``micro_autotune.py``) | 00:50.848 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_microtvm_micro_autotune.py` (``micro_autotune.py``) | 00:51.620 | 0.0 MB |
+---------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_microtvm_micro_aot.py` (``micro_aot.py``) | 00:07.681 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_microtvm_micro_aot.py` (``micro_aot.py``) | 00:07.914 | 0.0 MB |
+---------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_microtvm_micro_tflite.py` (``micro_tflite.py``) | 00:03.777 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_microtvm_micro_tflite.py` (``micro_tflite.py``) | 00:03.829 | 0.0 MB |
+---------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_work_with_microtvm_micro_reference_vm.py` (``micro_reference_vm.py``) | 00:00.001 | 0.0 MB |
+---------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/work_with_relay/sg_execution_times.rst.txt b/docs/_sources/how_to/work_with_relay/sg_execution_times.rst.txt
index cd1b78ce1e..38420d1abe 100644
--- a/docs/_sources/how_to/work_with_relay/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/work_with_relay/sg_execution_times.rst.txt
@@ -5,14 +5,14 @@
Computation times
=================
-**00:44.009** total execution time for **how_to_work_with_relay** files:
+**00:44.406** total execution time for **how_to_work_with_relay** files:
+----------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_relay_using_pipeline_executor.py` (``using_pipeline_executor.py``) | 00:32.270 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_relay_using_pipeline_executor.py` (``using_pipeline_executor.py``) | 00:32.573 | 0.0 MB |
+----------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_relay_using_external_lib.py` (``using_external_lib.py``) | 00:10.201 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_relay_using_external_lib.py` (``using_external_lib.py``) | 00:10.261 | 0.0 MB |
+----------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_relay_build_gcn.py` (``build_gcn.py``) | 00:01.531 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_relay_build_gcn.py` (``build_gcn.py``) | 00:01.565 | 0.0 MB |
+----------------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_work_with_relay_using_relay_viz.py` (``using_relay_viz.py``) | 00:00.007 | 0.0 MB |
+----------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/work_with_schedules/intrin_math.rst.txt b/docs/_sources/how_to/work_with_schedules/intrin_math.rst.txt
index 7a56d9c8f2..7171522975 100644
--- a/docs/_sources/how_to/work_with_schedules/intrin_math.rst.txt
+++ b/docs/_sources/how_to/work_with_schedules/intrin_math.rst.txt
@@ -261,7 +261,7 @@ The following example customizes CUDA lowering rule for :code:`exp`.
.. code-block:: none
- <function my_cuda_math_rule at 0x7ff67e92a320>
+ <function my_cuda_math_rule at 0x7f86122ac170>
diff --git a/docs/_sources/how_to/work_with_schedules/sg_execution_times.rst.txt b/docs/_sources/how_to/work_with_schedules/sg_execution_times.rst.txt
index 5880e73a56..1ed337410f 100644
--- a/docs/_sources/how_to/work_with_schedules/sg_execution_times.rst.txt
+++ b/docs/_sources/how_to/work_with_schedules/sg_execution_times.rst.txt
@@ -5,22 +5,22 @@
Computation times
=================
-**00:06.392** total execution time for **how_to_work_with_schedules** files:
+**00:08.291** total execution time for **how_to_work_with_schedules** files:
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_intrin_math.py` (``intrin_math.py``) | 00:03.827 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_intrin_math.py` (``intrin_math.py``) | 00:05.782 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_tensorize.py` (``tensorize.py``) | 00:01.214 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_tensorize.py` (``tensorize.py``) | 00:01.153 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_reduction.py` (``reduction.py``) | 00:00.578 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_reduction.py` (``reduction.py``) | 00:00.579 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_how_to_work_with_schedules_scan.py` (``scan.py``) | 00:00.556 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_extern_op.py` (``extern_op.py``) | 00:00.114 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_extern_op.py` (``extern_op.py``) | 00:00.115 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_schedule_primitives.py` (``schedule_primitives.py``) | 00:00.050 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_schedule_primitives.py` (``schedule_primitives.py``) | 00:00.052 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_tedd.py` (``tedd.py``) | 00:00.029 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_tedd.py` (``tedd.py``) | 00:00.030 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_how_to_work_with_schedules_tuple_inputs.py` (``tuple_inputs.py``) | 00:00.023 | 0.0 MB |
+| :ref:`sphx_glr_how_to_work_with_schedules_tuple_inputs.py` (``tuple_inputs.py``) | 00:00.025 | 0.0 MB |
+------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/how_to/work_with_schedules/tensorize.rst.txt b/docs/_sources/how_to/work_with_schedules/tensorize.rst.txt
index f705a20825..2c522945d2 100644
--- a/docs/_sources/how_to/work_with_schedules/tensorize.rst.txt
+++ b/docs/_sources/how_to/work_with_schedules/tensorize.rst.txt
@@ -343,7 +343,7 @@ The importing needs to happen before the tensorized GEMV being executed.
B: Buffer(B_2: Pointer(float32), float32, [512, 64], []),
C: Buffer(C_2: Pointer(float32), float32, [1024, 512], [])}
buffer_map = {A_1: A, B_1: B, C_1: C} {
- attr [IterVar(i: int32, (nullptr), "DataPar", "")] "pragma_import_llvm" = "; ModuleID = '/tmp/tmpcbhkriab/input0.cc'\nsource_filename = \"/tmp/tmpcbhkriab/input0.cc\"\ntarget datalayout = \"e-m:e-i64:64-f80:128-n8:16:32:64-S128\"\ntarget triple = \"x86_64-pc-linux-gnu\"\n\n; Function Attrs: noinline nounwind optnone uwtable\ndefine dso_local i32 @gemv_update(float*, float*, float*, i32, i32, i32) #0 {\n %7 = alloca float*, align 8\n %8 = alloca float*, align 8\n %9 = alloca floa [...]
+ attr [IterVar(i: int32, (nullptr), "DataPar", "")] "pragma_import_llvm" = "; ModuleID = '/tmp/tmp86lnfzzk/input0.cc'\nsource_filename = \"/tmp/tmp86lnfzzk/input0.cc\"\ntarget datalayout = \"e-m:e-i64:64-f80:128-n8:16:32:64-S128\"\ntarget triple = \"x86_64-pc-linux-gnu\"\n\n; Function Attrs: noinline nounwind optnone uwtable\ndefine dso_local i32 @gemv_update(float*, float*, float*, i32, i32, i32) #0 {\n %7 = alloca float*, align 8\n %8 = alloca float*, align 8\n %9 = alloca floa [...]
for (i, 0, 1024) {
for (j.outer: int32, 0, 32) {
@tir.call_extern("gemv_update", @tir.tvm_access_ptr(@tir.type_annotation(, dtype=float32), C_2, ((i*512) + (j.outer*16)), 16, 2, dtype=handle), @tir.tvm_access_ptr(@tir.type_annotation(, dtype=float32), A_2, (i*64), 64, 1, dtype=handle), @tir.tvm_access_ptr(@tir.type_annotation(, dtype=float32), B_2, (j.outer*1024), 1024, 1, dtype=handle), 16, 64, 64, dtype=int32)
diff --git a/docs/_sources/topic/vta/tutorials/autotvm/sg_execution_times.rst.txt b/docs/_sources/topic/vta/tutorials/autotvm/sg_execution_times.rst.txt
index ac26fc1249..6fb61132f2 100644
--- a/docs/_sources/topic/vta/tutorials/autotvm/sg_execution_times.rst.txt
+++ b/docs/_sources/topic/vta/tutorials/autotvm/sg_execution_times.rst.txt
@@ -5,10 +5,10 @@
Computation times
=================
-**00:26.295** total execution time for **topic_vta_tutorials_autotvm** files:
+**00:26.144** total execution time for **topic_vta_tutorials_autotvm** files:
+---------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_autotvm_tune_relay_vta.py` (``tune_relay_vta.py``) | 00:26.289 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_autotvm_tune_relay_vta.py` (``tune_relay_vta.py``) | 00:26.137 | 0.0 MB |
+---------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_topic_vta_tutorials_autotvm_tune_alu_vta.py` (``tune_alu_vta.py``) | 00:00.006 | 0.0 MB |
+---------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/topic/vta/tutorials/frontend/deploy_classification.rst.txt b/docs/_sources/topic/vta/tutorials/frontend/deploy_classification.rst.txt
index a09f4b44c6..1d1950d906 100644
--- a/docs/_sources/topic/vta/tutorials/frontend/deploy_classification.rst.txt
+++ b/docs/_sources/topic/vta/tutorials/frontend/deploy_classification.rst.txt
@@ -289,7 +289,7 @@ The compilation steps are:
DeprecationWarning,
/workspace/vta/tutorials/frontend/deploy_classification.py:213: DeprecationWarning: legacy graph executor behavior of producing json / lib / params will be removed in the next release. Please see documents of tvm.contrib.graph_executor.GraphModule for the new recommended usage.
relay_prog, target=tvm.target.Target(target, host=env.target_host), params=params
- resnet18_v1 inference graph built in 28.92s!
+ resnet18_v1 inference graph built in 29.15s!
diff --git a/docs/_sources/topic/vta/tutorials/frontend/deploy_detection.rst.txt b/docs/_sources/topic/vta/tutorials/frontend/deploy_detection.rst.txt
index 108a0f946a..24728e4591 100644
--- a/docs/_sources/topic/vta/tutorials/frontend/deploy_detection.rst.txt
+++ b/docs/_sources/topic/vta/tutorials/frontend/deploy_detection.rst.txt
@@ -333,7 +333,7 @@ The compilation steps are:
/workspace/python/tvm/relay/build_module.py:348: DeprecationWarning: Please use input parameter mod (tvm.IRModule) instead of deprecated parameter mod (tvm.relay.function.Function)
DeprecationWarning,
- yolov3-tiny inference graph built in 19.45s!
+ yolov3-tiny inference graph built in 19.76s!
diff --git a/docs/_sources/topic/vta/tutorials/frontend/sg_execution_times.rst.txt b/docs/_sources/topic/vta/tutorials/frontend/sg_execution_times.rst.txt
index db31837ca7..94b669bed4 100644
--- a/docs/_sources/topic/vta/tutorials/frontend/sg_execution_times.rst.txt
+++ b/docs/_sources/topic/vta/tutorials/frontend/sg_execution_times.rst.txt
@@ -5,10 +5,10 @@
Computation times
=================
-**01:40.199** total execution time for **topic_vta_tutorials_frontend** files:
+**01:40.672** total execution time for **topic_vta_tutorials_frontend** files:
+------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_frontend_deploy_detection.py` (``deploy_detection.py``) | 00:51.373 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_frontend_deploy_detection.py` (``deploy_detection.py``) | 00:51.588 | 0.0 MB |
+------------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_frontend_deploy_classification.py` (``deploy_classification.py``) | 00:48.826 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_frontend_deploy_classification.py` (``deploy_classification.py``) | 00:49.084 | 0.0 MB |
+------------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/topic/vta/tutorials/optimize/sg_execution_times.rst.txt b/docs/_sources/topic/vta/tutorials/optimize/sg_execution_times.rst.txt
index c41b218dcb..febb016eb3 100644
--- a/docs/_sources/topic/vta/tutorials/optimize/sg_execution_times.rst.txt
+++ b/docs/_sources/topic/vta/tutorials/optimize/sg_execution_times.rst.txt
@@ -5,10 +5,10 @@
Computation times
=================
-**00:03.181** total execution time for **topic_vta_tutorials_optimize** files:
+**00:03.145** total execution time for **topic_vta_tutorials_optimize** files:
+--------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_optimize_convolution_opt.py` (``convolution_opt.py``) | 00:02.725 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_optimize_convolution_opt.py` (``convolution_opt.py``) | 00:02.694 | 0.0 MB |
+--------------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_optimize_matrix_multiply_opt.py` (``matrix_multiply_opt.py``) | 00:00.456 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_optimize_matrix_multiply_opt.py` (``matrix_multiply_opt.py``) | 00:00.452 | 0.0 MB |
+--------------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/topic/vta/tutorials/sg_execution_times.rst.txt b/docs/_sources/topic/vta/tutorials/sg_execution_times.rst.txt
index 50b5725f8a..d34fa3c00e 100644
--- a/docs/_sources/topic/vta/tutorials/sg_execution_times.rst.txt
+++ b/docs/_sources/topic/vta/tutorials/sg_execution_times.rst.txt
@@ -5,10 +5,10 @@
Computation times
=================
-**00:00.795** total execution time for **topic_vta_tutorials** files:
+**00:00.801** total execution time for **topic_vta_tutorials** files:
+---------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_matrix_multiply.py` (``matrix_multiply.py``) | 00:00.423 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_matrix_multiply.py` (``matrix_multiply.py``) | 00:00.418 | 0.0 MB |
+---------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_topic_vta_tutorials_vta_get_started.py` (``vta_get_started.py``) | 00:00.372 | 0.0 MB |
+| :ref:`sphx_glr_topic_vta_tutorials_vta_get_started.py` (``vta_get_started.py``) | 00:00.383 | 0.0 MB |
+---------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/tutorial/auto_scheduler_matmul_x86.rst.txt b/docs/_sources/tutorial/auto_scheduler_matmul_x86.rst.txt
index bec1e044bd..f4947422fe 100644
--- a/docs/_sources/tutorial/auto_scheduler_matmul_x86.rst.txt
+++ b/docs/_sources/tutorial/auto_scheduler_matmul_x86.rst.txt
@@ -203,6 +203,13 @@ trials, we can load the best schedule from the log file and apply it.
+.. rst-class:: sphx-glr-script-out
+
+ .. code-block:: none
+
+ *E
+
+
@@ -325,7 +332,7 @@ We build the binary and check its correctness and performance.
.. code-block:: none
- Execution time of this operator: 95.310 ms
+ Execution time of this operator: 93.844 ms
@@ -425,7 +432,7 @@ resume the status and do more 5 trials.
Resume search:
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/xgboost/training.py:17: UserWarning: Old style callback is deprecated. See: https://xgboost.readthedocs.io/en/latest/python/callbacks.html
warnings.warn(f'Old style callback is deprecated. See: {link}', UserWarning)
-
+ *E
@@ -443,7 +450,7 @@ operations.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 1 minutes 17.168 seconds)
+ **Total running time of the script:** ( 1 minutes 33.745 seconds)
.. _sphx_glr_download_tutorial_auto_scheduler_matmul_x86.py:
diff --git a/docs/_sources/tutorial/autotvm_matmul_x86.rst.txt b/docs/_sources/tutorial/autotvm_matmul_x86.rst.txt
index ce5638dbbf..21668e7556 100644
--- a/docs/_sources/tutorial/autotvm_matmul_x86.rst.txt
+++ b/docs/_sources/tutorial/autotvm_matmul_x86.rst.txt
@@ -450,16 +450,16 @@ reduce variance, we take 5 measurements and average them.
waiting for device...
device available
Get devices for measurement successfully!
- No: 1 GFLOPS: 13.80/13.80 result: MeasureResult(costs=(0.0194558774,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.5803329944610596, timestamp=1670932108.9512684) [('tile_y', [-1, 128]), ('tile_x', [-1, 64])],None,67
- No: 2 GFLOPS: 3.27/13.80 result: MeasureResult(costs=(0.0820496946,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.5661671161651611, timestamp=1670932111.268001) [('tile_y', [-1, 32]), ('tile_x', [-1, 8])],None,35
- No: 3 GFLOPS: 1.25/13.80 result: MeasureResult(costs=(0.21448415920000002,), error_no=MeasureErrorNo.NO_ERROR, all_cost=3.6506502628326416, timestamp=1670932114.9510396) [('tile_y', [-1, 1]), ('tile_x', [-1, 2])],None,10
- No: 4 GFLOPS: 8.25/13.80 result: MeasureResult(costs=(0.032521541200000004,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.7481691837310791, timestamp=1670932116.4717638) [('tile_y', [-1, 512]), ('tile_x', [-1, 32])],None,59
- No: 5 GFLOPS: 10.80/13.80 result: MeasureResult(costs=(0.0248484722,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.6214299201965332, timestamp=1670932117.2302072) [('tile_y', [-1, 512]), ('tile_x', [-1, 512])],None,99
- No: 6 GFLOPS: 0.89/13.80 result: MeasureResult(costs=(0.30090856920000003,), error_no=MeasureErrorNo.NO_ERROR, all_cost=5.039398908615112, timestamp=1670932122.2939873) [('tile_y', [-1, 256]), ('tile_x', [-1, 2])],None,18
- No: 7 GFLOPS: 3.17/13.80 result: MeasureResult(costs=(0.0846517856,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.5900273323059082, timestamp=1670932124.6620882) [('tile_y', [-1, 2]), ('tile_x', [-1, 8])],None,31
- No: 8 GFLOPS: 0.90/13.80 result: MeasureResult(costs=(0.298070097,), error_no=MeasureErrorNo.NO_ERROR, all_cost=4.995677947998047, timestamp=1670932129.6812446) [('tile_y', [-1, 64]), ('tile_x', [-1, 2])],None,16
- No: 9 GFLOPS: 9.48/13.80 result: MeasureResult(costs=(0.028323294600000003,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.8889896869659424, timestamp=1670932130.685676) [('tile_y', [-1, 2]), ('tile_x', [-1, 128])],None,71
- No: 10 GFLOPS: 12.76/13.80 result: MeasureResult(costs=(0.0210358396,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.649216890335083, timestamp=1670932131.2805958) [('tile_y', [-1, 64]), ('tile_x', [-1, 128])],None,76
+ No: 1 GFLOPS: 13.07/13.07 result: MeasureResult(costs=(0.0205368184,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.5780963897705078, timestamp=1670948549.2738624) [('tile_y', [-1, 256]), ('tile_x', [-1, 64])],None,68
+ No: 2 GFLOPS: 3.87/13.07 result: MeasureResult(costs=(0.0693297656,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.374544620513916, timestamp=1670948550.6297207) [('tile_y', [-1, 32]), ('tile_x', [-1, 16])],None,45
+ No: 3 GFLOPS: 9.11/13.07 result: MeasureResult(costs=(0.029472160000000004,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.7829124927520752, timestamp=1670948552.1192386) [('tile_y', [-1, 1]), ('tile_x', [-1, 128])],None,70
+ No: 4 GFLOPS: 12.08/13.07 result: MeasureResult(costs=(0.0222229422,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.5970776081085205, timestamp=1670948553.494518) [('tile_y', [-1, 256]), ('tile_x', [-1, 256])],None,88
+ No: 5 GFLOPS: 3.27/13.07 result: MeasureResult(costs=(0.0820425618,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.5601208209991455, timestamp=1670948555.1876364) [('tile_y', [-1, 32]), ('tile_x', [-1, 8])],None,35
+ No: 6 GFLOPS: 13.78/13.78 result: MeasureResult(costs=(0.0194824548,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.5763728618621826, timestamp=1670948555.7506273) [('tile_y', [-1, 128]), ('tile_x', [-1, 64])],None,67
+ No: 7 GFLOPS: 12.50/13.78 result: MeasureResult(costs=(0.021482233400000002,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.6061413288116455, timestamp=1670948557.1230469) [('tile_y', [-1, 128]), ('tile_x', [-1, 256])],None,87
+ No: 8 GFLOPS: 3.46/13.78 result: MeasureResult(costs=(0.07759364960000001,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.4748997688293457, timestamp=1670948558.6127977) [('tile_y', [-1, 8]), ('tile_x', [-1, 8])],None,33
+ No: 9 GFLOPS: 11.44/13.78 result: MeasureResult(costs=(0.023467501999999998,), error_no=MeasureErrorNo.NO_ERROR, all_cost=0.6271872520446777, timestamp=1670948559.3747077) [('tile_y', [-1, 128]), ('tile_x', [-1, 32])],None,57
+ No: 10 GFLOPS: 3.90/13.78 result: MeasureResult(costs=(0.06885490200000001,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.3823633193969727, timestamp=1670948560.7359188) [('tile_y', [-1, 64]), ('tile_x', [-1, 16])],None,46
diff --git a/docs/_sources/tutorial/autotvm_relay_x86.rst.txt b/docs/_sources/tutorial/autotvm_relay_x86.rst.txt
index 8e102aac65..5b4db03764 100644
--- a/docs/_sources/tutorial/autotvm_relay_x86.rst.txt
+++ b/docs/_sources/tutorial/autotvm_relay_x86.rst.txt
@@ -320,7 +320,7 @@ standard deviation.
.. code-block:: none
- {'mean': 514.6123362599998, 'median': 514.5555499500006, 'std': 3.2917024785227706}
+ {'mean': 515.119266449999, 'median': 515.476666349997, 'std': 2.079642360524064}
@@ -554,30 +554,30 @@ the tuning data to.
.. code-block:: none
-
[Task 1/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 1/25] Current/Best: 7.52/ 23.33 GFLOPS | Progress: (4/20) | 8.38 s
[Task 1/25] Current/Best: 23.29/ 23.33 GFLOPS | Progress: (8/20) | 10.52 s
[Task 1/25] Current/Best: 5.71/ 23.33 GFLOPS | Progress: (12/20) | 13.15 s
[Task 1/25] Current/Best: 11.54/ 23.33 GFLOPS | Progress: (16/20) | 16.60 s
[Task 1/25] Current/Best: 14.99/ 23.33 GFLOPS | Progress: (20/20) | 20.27 s Done.
-
[Task 2/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 2/25] Current/Best: 11.35/ 16.05 GFLOPS | Progress: (4/20) | 3.41 s
[Task 2/25] Current/Best: 12.17/ 16.05 GFLOPS | Progress: (8/20) | 5.15 s
[Task 2/25] Current/Best: 19.26/ 19.26 GFLOPS | Progress: (12/20) | 6.69 s
[Task 2/25] Current/Best: 14.73/ 19.26 GFLOPS | Progress: (16/20) | 9.57 s
[Task 2/25] Current/Best: 18.32/ 21.00 GFLOPS | Progress: (20/20) | 11.13 s Done.
-
[Task 3/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 3/25] Current/Best: 13.45/ 14.82 GFLOPS | Progress: (4/20) | 4.76 s
[Task 3/25] Current/Best: 12.75/ 16.22 GFLOPS | Progress: (8/20) | 7.29 s
[Task 3/25] Current/Best: 15.89/ 16.22 GFLOPS | Progress: (12/20) | 10.24 s
[Task 3/25] Current/Best: 14.52/ 16.22 GFLOPS | Progress: (16/20) | 12.46 s
[Task 3/25] Current/Best: 6.26/ 17.66 GFLOPS | Progress: (20/20) | 15.13 s Done.
-
[Task 4/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 4/25] Current/Best: 5.54/ 23.28 GFLOPS | Progress: (4/20) | 3.82 s
[Task 4/25] Current/Best: 4.74/ 23.28 GFLOPS | Progress: (8/20) | 10.41 s
[Task 4/25] Current/Best: 13.96/ 23.28 GFLOPS | Progress: (12/20) | 12.48 s
[Task 4/25] Current/Best: 16.81/ 23.28 GFLOPS | Progress: (16/20) | 15.50 s
[Task 4/25] Current/Best: 10.24/ 23.28 GFLOPS | Progress: (20/20) | 17.44 s Done.
-
[Task 5/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 5/25] Current/Best: 17.67/ 17.67 GFLOPS | Progress: (4/20) | 3.72 s
[Task 5/25] Current/Best: 6.99/ 17.85 GFLOPS | Progress: (8/20) | 5.67 s
[Task 5/25] Current/Best: 5.99/ 17.85 GFLOPS | Progress: (12/20) | 7.90 s
[Task 5/25] Current/Best: 10.49/ 17.85 GFLOPS | Progress: (16/20) | 10.01 s
[Task 5/25] Current/Best: 2.95/ 18.01 GFLOPS | Progress: (20/20) | 12.23 s Done.
-
[Task 6/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 6/25] Current/Best: 12.23/ 14.89 GFLOPS | Progress: (4/20) | 5.66 s
[Task 6/25] Current/Best: 3.95/ 14.89 GFLOPS | Progress: (8/20) | 9.94 s
[Task 6/25] Current/Best: 21.44/ 21.44 GFLOPS | Progress: (12/20) | 12.19 s
[Task 6/25] Current/Best: 13.61/ 22.07 GFLOPS | Progress: (16/20) | 15.63 s
[Task 6/25] Current/Best: 5.07/ 22.07 GFLOPS | Progress: (20/20) | 18.21 s Done.
-
[Task 7/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 7/25] Current/Best: 11.41/ 18.43 GFLOPS | Progress: (4/20) | 4.47 s
[Task 7/25] Current/Best: 11.75/ 18.43 GFLOPS | Progress: (8/20) | 7.29 s
[Task 7/25] Current/Best: 6.31/ 18.43 GFLOPS | Progress: (12/20) | 10.39 s
[Task 7/25] Current/Best: 13.31/ 18.43 GFLOPS | Progress: (16/20) | 12.75 s
[Task 7/25] Current/Best: 13.50/ 20.52 GFLOPS | Progress: (20/20) | 14.72 s Done.
-
[Task 8/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 8/25] Current/Best: 11.74/ 11.74 GFLOPS | Progress: (4/20) | 9.04 s
[Task 8/25] Current/Best: 13.24/ 13.24 GFLOPS | Progress: (8/20) | 16.79 s
[Task 8/25] Current/Best: 13.65/ 13.65 GFLOPS | Progress: (12/20) | 19.96 s
[Task 8/25] Current/Best: 1.56/ 17.53 GFLOPS | Progress: (16/20) | 23.60 s
[Task 8/25] Current/Best: 13.40/ 17.53 GFLOPS | Progress: (20/20) | 27.99 s Done.
-
[Task 9/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 9/25] Current/Best: 9.96/ 13.07 GFLOPS | Progress: (4/20) | 6.29 s
[Task 9/25] Current/Best: 6.76/ 13.07 GFLOPS | Progress: (8/20) | 14.00 s
[Task 9/25] Current/Best: 14.76/ 18.77 GFLOPS | Progress: (12/20) | 17.71 s
[Task 9/25] Current/Best: 15.65/ 18.77 GFLOPS | Progress: (16/20) | 19.69 s
[Task 9/25] Current/Best: 7.05/ 21.32 GFLOPS | Progress: (20/20) | 24.83 s Done.
-
[Task 10/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 10/25] Current/Best: 4.04/ 13.65 GFLOPS | Progress: (4/20) | 3.99 s
[Task 10/25] Current/Best: 13.93/ 13.93 GFLOPS | Progress: (8/20) | 6.14 s
[Task 10/25] Current/Best: 15.59/ 17.68 GFLOPS | Progress: (12/20) | 7.90 s
[Task 10/25] Current/Best: 9.07/ 18.83 GFLOPS | Progress: (16/20) | 9.93 s
[Task 10/25] Current/Best: 4.74/ 18.83 GFLOPS | Progress: (20/20) | 12.17 s Done.
-
[Task 11/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 11/25] Current/Best: 11.89/ 23.10 GFLOPS | Progress: (4/20) | 5.35 s
[Task 11/25] Current/Best: 6.02/ 23.65 GFLOPS | Progress: (8/20) | 7.59 s
[Task 11/25] Current/Best: 11.16/ 23.65 GFLOPS | Progress: (12/20) | 10.09 s
[Task 11/25] Current/Best: 4.57/ 23.65 GFLOPS | Progress: (16/20) | 12.52 s
[Task 11/25] Current/Best: 12.23/ 23.65 GFLOPS | Progress: (20/20) | 15.88 s Done.
-
[Task 12/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 12/25] Current/Best: 4.66/ 16.55 GFLOPS | Progress: (4/20) | 4.08 s
[Task 12/25] Current/Best: 9.10/ 16.55 GFLOPS | Progress: (8/20) | 7.19 s
[Task 12/25] Current/Best: 6.43/ 16.55 GFLOPS | Progress: (12/20) | 11.23 s
[Task 12/25] Current/Best: 12.34/ 16.56 GFLOPS | Progress: (16/20) | 16.62 s
[Task 12/25] Current/Best: 13.04/ 17.38 GFLOPS | Progress: (20/20) | 18.61 s Done.
-
[Task 13/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 13/25] Current/Best: 8.35/ 17.66 GFLOPS | Progress: (4/20) | 4.74 s
[Task 13/25] Current/Best: 17.01/ 21.42 GFLOPS | Progress: (8/20) | 7.63 s
[Task 13/25] Current/Best: 16.56/ 21.42 GFLOPS | Progress: (12/20) | 9.85 s
[Task 13/25] Current/Best: 13.53/ 22.04 GFLOPS | Progress: (16/20) | 12.82 s
[Task 13/25] Current/Best: 12.73/ 22.04 GFLOPS | Progress: (20/20) | 15.43 s Done.
-
[Task 14/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 14/25] Current/Best: 18.35/ 18.35 GFLOPS | Progress: (4/20) | 3.54 s
[Task 14/25] Current/Best: 7.15/ 18.35 GFLOPS | Progress: (8/20) | 7.26 s
[Task 14/25] Current/Best: 8.01/ 18.35 GFLOPS | Progress: (12/20) | 9.58 s
[Task 14/25] Current/Best: 11.86/ 18.35 GFLOPS | Progress: (16/20) | 12.52 s
[Task 14/25] Current/Best: 12.83/ 18.35 GFLOPS | Progress: (20/20) | 15.66 s
[Task 15/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 15/25] Current/Best: 3.24/ 15.08 GFLOPS | Progress: (4/20) | 4.45 s
[Task 15/25] Current/Best: 9.65/ 19.44 GFLOPS | Progress: (8/20) | 7.28 s
[Task 15/25] Current/Best: 6.03/ 19.44 GFLOPS | Progress: (12/20) | 8.97 s Done.
-
[Task 15/25] Current/Best: 20.35/ 20.35 GFLOPS | Progress: (16/20) | 13.25 s
[Task 15/25] Current/Best: 10.25/ 20.93 GFLOPS | Progress: (20/20) | 15.42 s Done.
-
[Task 16/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 16/25] Current/Best: 11.42/ 12.24 GFLOPS | Progress: (4/20) | 4.90 s
[Task 16/25] Current/Best: 7.58/ 16.24 GFLOPS | Progress: (8/20) | 6.93 s
[Task 16/25] Current/Best: 17.10/ 20.25 GFLOPS | Progress: (12/20) | 8.51 s
[Task 16/25] Current/Best: 9.11/ 20.25 GFLOPS | Progress: (16/20) | 10.66 s
[Task 16/25] Current/Best: 11.53/ 20.25 GFLOPS | Progress: (20/20) | 13.37 s Done.
-
[Task 17/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 17/25] Current/Best: 6.20/ 18.46 GFLOPS | Progress: (4/20) | 4.92 s
[Task 17/25] Current/Best: 11.32/ 19.81 GFLOPS | Progress: (8/20) | 7.62 s
[Task 17/25] Current/Best: 11.33/ 21.80 GFLOPS | Progress: (12/20) | 11.52 s
[Task 17/25] Current/Best: 19.78/ 21.80 GFLOPS | Progress: (16/20) | 13.90 s
[Task 17/25] Current/Best: 12.93/ 21.80 GFLOPS | Progress: (20/20) | 16.96 s Done.
-
[Task 18/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 18/25] Current/Best: 15.26/ 18.41 GFLOPS | Progress: (4/20) | 7.43 s
[Task 18/25] Current/Best: 16.66/ 19.27 GFLOPS | Progress: (8/20) | 9.88 s
[Task 18/25] Current/Best: 6.01/ 19.27 GFLOPS | Progress: (12/20) | 13.43 s
[Task 18/25] Current/Best: 9.99/ 19.27 GFLOPS | Progress: (16/20) | 15.49 s
[Task 18/25] Current/Best: 10.76/ 19.27 GFLOPS | Progress: (20/20) | 19.60 s Done.
-
[Task 19/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 19/25] Current/Best: 17.07/ 17.07 GFLOPS | Progress: (4/20) | 5.37 s
[Task 19/25] Current/Best: 17.59/ 19.02 GFLOPS | Progress: (8/20) | 7.96 s
[Task 19/25] Current/Best: 3.07/ 19.02 GFLOPS | Progress: (12/20) | 12.01 s
[Task 19/25] Current/Best: 5.34/ 19.02 GFLOPS | Progress: (16/20) | 15.41 s
[Task 19/25] Current/Best: 18.61/ 19.02 GFLOPS | Progress: (20/20) | 19.05 s Done.
-
[Task 20/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 20/25] Current/Best: 14.52/ 14.52 GFLOPS | Progress: (4/20) | 3.47 s
[Task 20/25] Current/Best: 4.55/ 14.52 GFLOPS | Progress: (8/20) | 6.77 s
[Task 20/25] Current/Best: 16.83/ 16.83 GFLOPS | Progress: (12/20) | 9.99 s
[Task 20/25] Current/Best: 6.46/ 16.83 GFLOPS | Progress: (16/20) | 12.47 s
[Task 20/25] Current/Best: 5.39/ 16.83 GFLOPS | Progress: (20/20) | 16.48 s
[Task 21/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 21/25] Current/Best: 2.75/ 16.51 GFLOPS | Progress: (4/20) | 9.43 s Done.
-
[Task 21/25] Current/Best: 13.08/ 17.69 GFLOPS | Progress: (8/20) | 10.96 s
[Task 21/25] Current/Best: 19.92/ 19.92 GFLOPS | Progress: (12/20) | 13.17 s
[Task 21/25] Current/Best: 10.67/ 19.92 GFLOPS | Progress: (16/20) | 15.13 s
[Task 21/25] Current/Best: 10.67/ 19.92 GFLOPS | Progress: (20/20) | 17.94 s
[Task 22/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 22/25] Current/Best: 6.76/ 17.70 GFLOPS | Progress: (4/20) | 4.18 s
[Task 22/25] Current/Best: 5.44/ 18.48 GFLOPS | Progress: (8/20) | 7.41 s
[Task 22/25] Current/Best: 13.33/ 18.48 GFLOPS | Progress: (12/20) | 9.08 s
[Task 22/25] Current/Best: 16.47/ 18.48 GFLOPS | Progress: (16/20) | 11.00 s
[Task 22/25] Current/Best: 16.22/ 18.48 GFLOPS | Progress: (20/20) | 12.85 s Done.
-
[Task 23/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 23/25] Current/Best: 1.55/ 20.62 GFLOPS | Progress: (4/20) | 5.74 s
[Task 23/25] Current/Best: 16.35/ 20.62 GFLOPS | Progress: (8/20) | 8.36 s
[Task 23/25] Current/Best: 7.87/ 20.62 GFLOPS | Progress: (12/20) | 13.91 s
[Task 23/25] Current/Best: 14.14/ 20.62 GFLOPS | Progress: (16/20) | 16.52 s
[Task 23/25] Current/Best: 2.69/ 22.41 GFLOPS | Progress: (20/20) | 19.85 s Done.
-
[Task 24/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 24/25] Current/Best: 7.52/ 7.52 GFLOPS | Progress: (4/20) | 8.50 s
[Task 24/25] Current/Best: 3.67/ 7.93 GFLOPS | Progress: (8/20) | 19.43 s
[Task 24/25] Current/Best: 1.88/ 9.11 GFLOPS | Progress: (12/20) | 25.18 s
[Task 24/25] Current/Best: 3.08/ 9.11 GFLOPS | Progress: (16/20) | 26.63 s
[Task 24/25] Current/Best: 6.76/ 9.11 GFLOPS | Progress: (20/20) | 37.27 s
[Task 25/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s Done.
-
[Task 25/25] Current/Best: 3.45/ 8.78 GFLOPS | Progress: (4/20) | 12.71 s
[Task 25/25] Current/Best: 9.02/ 9.02 GFLOPS | Progress: (8/20) | 23.62 s
[Task 25/25] Current/Best: 8.18/ 9.02 GFLOPS | Progress: (12/20) | 26.94 s
[Task 25/25] Current/Best: 3.50/ 9.02 GFLOPS | Progress: (16/20) | 37.91 s
[Task 25/25] Current/Best: 3.03/ 9.02 GFLOPS | Progress: (20/20) | 39.98 s
+
[Task 1/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 1/25] Current/Best: 9.54/ 23.24 GFLOPS | Progress: (4/20) | 7.76 s
[Task 1/25] Current/Best: 11.11/ 23.24 GFLOPS | Progress: (8/20) | 11.82 s
[Task 1/25] Current/Best: 11.21/ 23.24 GFLOPS | Progress: (12/20) | 14.67 s
[Task 1/25] Current/Best: 21.78/ 23.24 GFLOPS | Progress: (16/20) | 17.90 s
[Task 1/25] Current/Best: 12.57/ 23.24 GFLOPS | Progress: (20/20) | 20.25 s Done.
+
[Task 2/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 2/25] Current/Best: 6.13/ 17.13 GFLOPS | Progress: (4/20) | 3.45 s
[Task 2/25] Current/Best: 11.49/ 22.15 GFLOPS | Progress: (8/20) | 6.18 s
[Task 2/25] Current/Best: 18.16/ 22.15 GFLOPS | Progress: (12/20) | 7.84 s
[Task 2/25] Current/Best: 11.73/ 22.15 GFLOPS | Progress: (16/20) | 9.70 s
[Task 2/25] Current/Best: 11.40/ 22.15 GFLOPS | Progress: (20/20) | 11.37 s Done.
+
[Task 3/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 3/25] Current/Best: 20.25/ 20.25 GFLOPS | Progress: (4/20) | 3.96 s
[Task 3/25] Current/Best: 15.86/ 20.25 GFLOPS | Progress: (8/20) | 6.68 s
[Task 3/25] Current/Best: 18.77/ 20.25 GFLOPS | Progress: (12/20) | 9.18 s
[Task 3/25] Current/Best: 11.36/ 20.25 GFLOPS | Progress: (16/20) | 11.73 s
[Task 3/25] Current/Best: 12.68/ 23.11 GFLOPS | Progress: (20/20) | 13.92 s Done.
+
[Task 4/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 4/25] Current/Best: 10.08/ 12.41 GFLOPS | Progress: (4/20) | 6.76 s
[Task 4/25] Current/Best: 12.43/ 13.34 GFLOPS | Progress: (8/20) | 10.12 s
[Task 4/25] Current/Best: 11.23/ 21.39 GFLOPS | Progress: (12/20) | 11.98 s
[Task 4/25] Current/Best: 13.96/ 21.39 GFLOPS | Progress: (16/20) | 14.48 s
[Task 4/25] Current/Best: 14.32/ 21.39 GFLOPS | Progress: (20/20) | 17.03 s Done.
+
[Task 5/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 5/25] Current/Best: 12.17/ 17.77 GFLOPS | Progress: (4/20) | 3.57 s
[Task 5/25] Current/Best: 9.85/ 22.35 GFLOPS | Progress: (8/20) | 5.28 s
[Task 5/25] Current/Best: 12.81/ 22.35 GFLOPS | Progress: (12/20) | 9.02 s
[Task 5/25] Current/Best: 9.51/ 22.35 GFLOPS | Progress: (16/20) | 11.26 s
[Task 5/25] Current/Best: 4.31/ 22.35 GFLOPS | Progress: (20/20) | 13.74 s Done.
+
[Task 6/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 6/25] Current/Best: 17.47/ 17.75 GFLOPS | Progress: (4/20) | 4.38 s
[Task 6/25] Current/Best: 4.72/ 17.75 GFLOPS | Progress: (8/20) | 7.94 s
[Task 6/25] Current/Best: 5.45/ 20.61 GFLOPS | Progress: (12/20) | 10.25 s
[Task 6/25] Current/Best: 14.51/ 21.70 GFLOPS | Progress: (16/20) | 12.58 s
[Task 6/25] Current/Best: 18.22/ 21.70 GFLOPS | Progress: (20/20) | 15.07 s Done.
+
[Task 7/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 7/25] Current/Best: 13.41/ 16.81 GFLOPS | Progress: (4/20) | 4.02 s
[Task 7/25] Current/Best: 12.11/ 20.01 GFLOPS | Progress: (8/20) | 6.80 s
[Task 7/25] Current/Best: 15.56/ 20.01 GFLOPS | Progress: (12/20) | 9.20 s
[Task 7/25] Current/Best: 12.00/ 20.01 GFLOPS | Progress: (16/20) | 12.15 s
[Task 7/25] Current/Best: 8.05/ 20.01 GFLOPS | Progress: (20/20) | 14.49 s Done.
+
[Task 8/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 8/25] Current/Best: 12.96/ 12.96 GFLOPS | Progress: (4/20) | 7.55 s
[Task 8/25] Current/Best: 4.76/ 22.10 GFLOPS | Progress: (8/20) | 19.06 s
[Task 8/25] Current/Best: 19.51/ 22.10 GFLOPS | Progress: (12/20) | 22.22 s
[Task 8/25] Current/Best: 11.13/ 22.10 GFLOPS | Progress: (16/20) | 25.10 s
[Task 8/25] Current/Best: 9.57/ 22.10 GFLOPS | Progress: (20/20) | 37.15 s
[Task 9/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 9/25] Current/Best: 20.20/ 20.20 GFLOPS | Progress: (4/20) | 12.94 s
[Task 9/25] Current/Best: 10.19/ 20.20 GFLOPS | Progress: (8/20) | 17.40 s
[Task 9/25] Current/Best: 12.16/ 20.20 GFLOPS | Progress: (12/20) | 19.41 s
[Task 9/25] Current/Best: 11.06/ 20.20 GFLOPS | Progress: (16/20) | 24.29 s
[Task 9/25] Current/Best: 9.61/ 20.20 GFLOPS | Progress: (20
/20) | 28.30 s
[Task 10/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 10/25] Current/Best: 19.03/ 20.02 GFLOPS | Progress: (4/20) | 3.65 s
[Task 10/25] Current/Best: 8.61/ 20.02 GFLOPS | Progress: (8/20) | 7.07 s
[Task 10/25] Current/Best: 18.40/ 20.02 GFLOPS | Progress: (12/20) | 9.32 s
[Task 10/25] Current/Best: 15.73/ 20.02 GFLOPS | Progress: (16/20) | 11.77 s
[Task 10/25] Current/Best: 12.93/ 20.02 GFLOPS | Progress: (20/20) | 14.56 s Done.
+
[Task 11/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 11/25] Current/Best: 14.61/ 19.38 GFLOPS | Progress: (4/20) | 4.78 s
[Task 11/25] Current/Best: 6.08/ 19.38 GFLOPS | Progress: (8/20) | 7.31 s
[Task 11/25] Current/Best: 9.44/ 19.38 GFLOPS | Progress: (12/20) | 9.68 s
[Task 11/25] Current/Best: 12.30/ 19.38 GFLOPS | Progress: (16/20) | 11.90 s
[Task 11/25] Current/Best: 12.19/ 19.38 GFLOPS | Progress: (20/20) | 15.57 s Done.
+
[Task 12/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 12/25] Current/Best: 8.19/ 8.19 GFLOPS | Progress: (4/20) | 6.22 s
[Task 12/25] Current/Best: 7.77/ 14.79 GFLOPS | Progress: (8/20) | 9.00 s
[Task 12/25] Current/Best: 19.14/ 19.14 GFLOPS | Progress: (12/20) | 11.79 s
[Task 12/25] Current/Best: 5.86/ 20.50 GFLOPS | Progress: (16/20) | 14.46 s
[Task 12/25] Current/Best: 15.23/ 20.50 GFLOPS | Progress: (20/20) | 17.96 s Done.
+
[Task 13/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 13/25] Current/Best: 15.22/ 15.22 GFLOPS | Progress: (4/20) | 4.97 s
[Task 13/25] Current/Best: 11.63/ 15.22 GFLOPS | Progress: (8/20) | 7.56 s
[Task 13/25] Current/Best: 13.04/ 18.99 GFLOPS | Progress: (12/20) | 9.82 s
[Task 13/25] Current/Best: 14.57/ 22.86 GFLOPS | Progress: (16/20) | 13.52 s
[Task 13/25] Current/Best: 16.68/ 22.86 GFLOPS | Progress: (20/20) | 16.53 s Done.
+
[Task 14/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 14/25] Current/Best: 13.46/ 18.22 GFLOPS | Progress: (4/20) | 4.85 s
[Task 14/25] Current/Best: 5.10/ 18.22 GFLOPS | Progress: (8/20) | 10.81 s Done.
+ Done.
+
[Task 14/25] Current/Best: 6.40/ 18.22 GFLOPS | Progress: (12/20) | 14.47 s
[Task 14/25] Current/Best: 15.50/ 18.22 GFLOPS | Progress: (16/20) | 16.58 s
[Task 14/25] Current/Best: 11.80/ 18.22 GFLOPS | Progress: (20/20) | 20.39 s Done.
+
[Task 15/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 15/25] Current/Best: 10.79/ 22.83 GFLOPS | Progress: (4/20) | 3.71 s
[Task 15/25] Current/Best: 14.19/ 22.83 GFLOPS | Progress: (8/20) | 6.50 s
[Task 15/25] Current/Best: 12.46/ 22.83 GFLOPS | Progress: (12/20) | 8.55 s
[Task 15/25] Current/Best: 21.51/ 22.83 GFLOPS | Progress: (16/20) | 10.45 s
[Task 15/25] Current/Best: 22.41/ 22.83 GFLOPS | Progress: (20/20) | 13.31 s
[Task 16/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 16/25] Current/Best: 15.40/ 20.64 GFLOPS | Progress: (4/20) | 4.13 s
[Task 16/25] Current/Best: 19.06/ 20.64 GFLOPS | Progress: (8/20) | 6.87 s
[Task 16/25] Current/Best: 13.01/ 20.64 GFLOPS | Progress: (12/20) | 8.79 s
[Task 16/25] Current/Best: 5.12/ 20.64 GFLOPS | Progress: (16/20) | 10.97 s
[Task 16/25] Current/Best: 15.76/ 20.64 GFLOPS | Progress: (20/20)
| 12.72 s Done.
+
[Task 17/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 17/25] Current/Best: 5.08/ 20.54 GFLOPS | Progress: (4/20) | 5.02 s
[Task 17/25] Current/Best: 20.89/ 20.89 GFLOPS | Progress: (8/20) | 8.20 s
[Task 17/25] Current/Best: 11.77/ 21.99 GFLOPS | Progress: (12/20) | 12.12 s
[Task 17/25] Current/Best: 17.86/ 21.99 GFLOPS | Progress: (16/20) | 14.87 s
[Task 17/25] Current/Best: 16.89/ 21.99 GFLOPS | Progress: (20/20) | 17.11 s Done.
+
[Task 18/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 18/25] Current/Best: 17.16/ 17.16 GFLOPS | Progress: (4/20) | 5.32 s
[Task 18/25] Current/Best: 1.58/ 17.16 GFLOPS | Progress: (8/20) | 10.10 s
[Task 18/25] Current/Best: 10.57/ 17.16 GFLOPS | Progress: (12/20) | 13.04 s
[Task 18/25] Current/Best: 14.74/ 19.52 GFLOPS | Progress: (16/20) | 20.25 s
[Task 18/25] Current/Best: 16.07/ 19.52 GFLOPS | Progress: (20/20) | 22.45 s Done.
+
[Task 19/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 19/25] Current/Best: 10.94/ 17.99 GFLOPS | Progress: (4/20) | 5.43 s
[Task 19/25] Current/Best: 16.06/ 17.99 GFLOPS | Progress: (8/20) | 9.54 s
[Task 19/25] Current/Best: 11.96/ 19.20 GFLOPS | Progress: (12/20) | 12.30 s
[Task 19/25] Current/Best: 5.27/ 19.20 GFLOPS | Progress: (16/20) | 15.60 s
[Task 19/25] Current/Best: 9.61/ 19.20 GFLOPS | Progress: (20/20) | 18.73 s Done.
+
[Task 20/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 20/25] Current/Best: 10.49/ 13.32 GFLOPS | Progress: (4/20) | 4.29 s
[Task 20/25] Current/Best: 12.69/ 13.32 GFLOPS | Progress: (8/20) | 6.97 s
[Task 20/25] Current/Best: 6.65/ 20.16 GFLOPS | Progress: (12/20) | 9.78 s
[Task 20/25] Current/Best: 5.14/ 20.16 GFLOPS | Progress: (16/20) | 11.84 s
[Task 20/25] Current/Best: 9.13/ 20.16 GFLOPS | Progress: (20/20) | 14.36 s
[Task 21/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s Done.
+ Done.
+
[Task 21/25] Current/Best: 9.58/ 17.05 GFLOPS | Progress: (4/20) | 3.86 s
[Task 21/25] Current/Best: 17.95/ 17.95 GFLOPS | Progress: (8/20) | 5.71 s
[Task 21/25] Current/Best: 18.39/ 18.39 GFLOPS | Progress: (12/20) | 8.72 s
[Task 21/25] Current/Best: 14.63/ 18.39 GFLOPS | Progress: (16/20) | 10.61 s
[Task 21/25] Current/Best: 9.77/ 18.39 GFLOPS | Progress: (20/20) | 13.01 s
[Task 22/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 22/25] Current/Best: 18.65/ 21.61 GFLOPS | Progress: (4/20) | 4.13 s
[Task 22/25] Current/Best: 17.61/ 21.61 GFLOPS | Progress: (8/20) | 6.39 s
[Task 22/25] Current/Best: 11.41/ 21.61 GFLOPS | Progress: (12/20) | 8.15 s
[Task 22/25] Current/Best: 17.72/ 21.61 GFLOPS | Progress: (16/20) | 10.11 s
[Task 22/25] Current/Best: 16.08/ 21.61 GFLOPS | Progress: (20/20) | 11.94 s Done.
+
[Task 23/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 23/25] Current/Best: 5.32/ 17.89 GFLOPS | Progress: (4/20) | 4.70 s
[Task 23/25] Current/Best: 8.99/ 21.53 GFLOPS | Progress: (8/20) | 6.94 s
[Task 23/25] Current/Best: 10.60/ 21.53 GFLOPS | Progress: (12/20) | 10.93 s
[Task 23/25] Current/Best: 19.58/ 21.53 GFLOPS | Progress: (16/20) | 13.96 s
[Task 23/25] Current/Best: 15.79/ 21.53 GFLOPS | Progress: (20/20) | 17.00 s Done.
+
[Task 24/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s
[Task 24/25] Current/Best: 5.25/ 10.11 GFLOPS | Progress: (4/20) | 12.71 s
[Task 24/25] Current/Best: 6.37/ 10.11 GFLOPS | Progress: (8/20) | 20.00 s
[Task 24/25] Current/Best: 2.98/ 10.11 GFLOPS | Progress: (12/20) | 30.37 s
[Task 24/25] Current/Best: 7.64/ 10.11 GFLOPS | Progress: (16/20) | 42.22 s
[Task 24/25] Current/Best: 5.14/ 10.11 GFLOPS | Progress: (20/20) | 54.07 s
[Task 25/25] Current/Best: 0.00/ 0.00 GFLOPS | Progress: (0/20) | 0.00 s Done.
+
[Task 25/25] Current/Best: 5.78/ 8.97 GFLOPS | Progress: (4/20) | 12.77 s
[Task 25/25] Current/Best: 6.01/ 9.49 GFLOPS | Progress: (8/20) | 23.71 s
[Task 25/25] Current/Best: 1.54/ 9.49 GFLOPS | Progress: (12/20) | 25.83 s
[Task 25/25] Current/Best: 5.08/ 9.49 GFLOPS | Progress: (16/20) | 36.78 s
[Task 25/25] Current/Best: 2.73/ 9.49 GFLOPS | Progress: (20/20) | 39.01 s
@@ -674,7 +674,7 @@ Verify that the optimized model runs and produces the same results:
.. code-block:: none
class='n02123045 tabby, tabby cat' with probability=0.621104
- class='n02123159 tiger cat' with probability=0.356378
+ class='n02123159 tiger cat' with probability=0.356379
class='n02124075 Egyptian cat' with probability=0.019712
class='n02129604 tiger, Panthera tigris' with probability=0.001215
class='n04040759 radiator' with probability=0.000262
@@ -731,8 +731,8 @@ improvement in comparing the optimized model to the unoptimized model.
.. code-block:: none
- optimized: {'mean': 408.52393536000363, 'median': 408.6265036499981, 'std': 0.4208245868000364}
- unoptimized: {'mean': 514.6123362599998, 'median': 514.5555499500006, 'std': 3.2917024785227706}
+ optimized: {'mean': 416.8094780100023, 'median': 416.87683979999974, 'std': 1.3962309249037268}
+ unoptimized: {'mean': 515.119266449999, 'median': 515.476666349997, 'std': 2.079642360524064}
@@ -755,7 +755,7 @@ profiling/benchmarking.
.. rst-class:: sphx-glr-timing
- **Total running time of the script:** ( 11 minutes 15.515 seconds)
+ **Total running time of the script:** ( 11 minutes 37.496 seconds)
.. _sphx_glr_download_tutorial_autotvm_relay_x86.py:
diff --git a/docs/_sources/tutorial/cross_compilation_and_rpc.rst.txt b/docs/_sources/tutorial/cross_compilation_and_rpc.rst.txt
index 36390adf7f..19522769aa 100644
--- a/docs/_sources/tutorial/cross_compilation_and_rpc.rst.txt
+++ b/docs/_sources/tutorial/cross_compilation_and_rpc.rst.txt
@@ -270,7 +270,7 @@ device and returns the measured cost. Network overhead is excluded.
.. code-block:: none
- 1.249e-07 secs/op
+ 1.43e-07 secs/op
diff --git a/docs/_sources/tutorial/intro_topi.rst.txt b/docs/_sources/tutorial/intro_topi.rst.txt
index bca81c40ed..170135802b 100644
--- a/docs/_sources/tutorial/intro_topi.rst.txt
+++ b/docs/_sources/tutorial/intro_topi.rst.txt
@@ -260,7 +260,7 @@ As you can see, scheduled stages of computation have been accumulated and we can
.. code-block:: none
- [stage(a, placeholder(a, 0x2258a950)), stage(b, placeholder(b, 0x22585680)), stage(T_add, compute(T_add, body=[(a[ax0, ax1, ax2] + b[ax1, ax2])], axis=[iter_var(ax0, range(min=0, ext=100)), iter_var(ax1, range(min=0, ext=10)), iter_var(ax2, range(min=0, ext=10))], reduce_axis=[], tag=broadcast, attrs={})), stage(T_multiply, compute(T_multiply, body=[(a[ax0, ax1, ax2]*b[ax1, ax2])], axis=[iter_var(ax0, range(min=0, ext=100)), iter_var(ax1, range(min=0, ext=10)), iter_var(ax2, range(mi [...]
+ [stage(a, placeholder(a, 0x20f6ca30)), stage(b, placeholder(b, 0x5457ad0)), stage(T_add, compute(T_add, body=[(a[ax0, ax1, ax2] + b[ax1, ax2])], axis=[iter_var(ax0, range(min=0, ext=100)), iter_var(ax1, range(min=0, ext=10)), iter_var(ax2, range(min=0, ext=10))], reduce_axis=[], tag=broadcast, attrs={})), stage(T_multiply, compute(T_multiply, body=[(a[ax0, ax1, ax2]*b[ax1, ax2])], axis=[iter_var(ax0, range(min=0, ext=100)), iter_var(ax1, range(min=0, ext=10)), iter_var(ax2, range(min [...]
diff --git a/docs/_sources/tutorial/sg_execution_times.rst.txt b/docs/_sources/tutorial/sg_execution_times.rst.txt
index e56e13ac1f..d5180314a1 100644
--- a/docs/_sources/tutorial/sg_execution_times.rst.txt
+++ b/docs/_sources/tutorial/sg_execution_times.rst.txt
@@ -5,28 +5,28 @@
Computation times
=================
-**14:36.795** total execution time for **tutorial** files:
+**15:02.645** total execution time for **tutorial** files:
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_autotvm_relay_x86.py` (``autotvm_relay_x86.py``) | 11:15.515 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_autotvm_relay_x86.py` (``autotvm_relay_x86.py``) | 11:37.496 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_auto_scheduler_matmul_x86.py` (``auto_scheduler_matmul_x86.py``) | 01:17.168 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_auto_scheduler_matmul_x86.py` (``auto_scheduler_matmul_x86.py``) | 01:33.745 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_tensor_expr_get_started.py` (``tensor_expr_get_started.py``) | 01:00.931 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_tensor_expr_get_started.py` (``tensor_expr_get_started.py``) | 00:58.686 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_relay_quick_start.py` (``relay_quick_start.py``) | 00:33.670 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_relay_quick_start.py` (``relay_quick_start.py``) | 00:33.804 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_autotvm_matmul_x86.py` (``autotvm_matmul_x86.py``) | 00:27.044 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_autotvm_matmul_x86.py` (``autotvm_matmul_x86.py``) | 00:16.333 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_tensor_ir_blitz_course.py` (``tensor_ir_blitz_course.py``) | 00:01.470 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_tensor_ir_blitz_course.py` (``tensor_ir_blitz_course.py``) | 00:01.588 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_intro_topi.py` (``intro_topi.py``) | 00:00.817 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_intro_topi.py` (``intro_topi.py``) | 00:00.816 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_cross_compilation_and_rpc.py` (``cross_compilation_and_rpc.py``) | 00:00.170 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_cross_compilation_and_rpc.py` (``cross_compilation_and_rpc.py``) | 00:00.168 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_tutorial_introduction.py` (``introduction.py``) | 00:00.007 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
-| :ref:`sphx_glr_tutorial_uma.py` (``uma.py``) | 00:00.002 | 0.0 MB |
+| :ref:`sphx_glr_tutorial_uma.py` (``uma.py``) | 00:00.001 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
| :ref:`sphx_glr_tutorial_tvmc_python.py` (``tvmc_python.py``) | 00:00.001 | 0.0 MB |
+------------------------------------------------------------------------------------------+-----------+--------+
diff --git a/docs/_sources/tutorial/tensor_expr_get_started.rst.txt b/docs/_sources/tutorial/tensor_expr_get_started.rst.txt
index 046d761f61..1ff779178b 100644
--- a/docs/_sources/tutorial/tensor_expr_get_started.rst.txt
+++ b/docs/_sources/tutorial/tensor_expr_get_started.rst.txt
@@ -294,8 +294,8 @@ helper function to run a profile of the TVM generated code.
.. code-block:: none
- Numpy running time: 0.000012
- naive: 0.000011
+ Numpy running time: 0.000007
+ naive: 0.000007
@@ -393,7 +393,7 @@ compile and run this new schedule with the parallel operation applied:
.. code-block:: none
- parallel: 0.000009
+ parallel: 0.000007
@@ -499,10 +499,10 @@ We can now compare the different schedules
.. code-block:: none
Operator Timing Performance
- numpy 1.1964479999733158e-05 1.0
- naive 1.1294300000000001e-05 0.9439858648476069
- parallel 9.027999999999999e-06 0.7545668512297524
- vector 2.46138e-05 2.057239428754861
+ numpy 6.636129999151308e-06 1.0
+ naive 6.7299e-06 1.0141302236183867
+ parallel 6.8983e-06 1.039506459469935
+ vector 2.47125e-05 3.723932473167414
@@ -923,7 +923,7 @@ matrix multiplication.
.. code-block:: none
- Numpy running time: 0.018027
+ Numpy running time: 0.018389
@@ -981,7 +981,7 @@ optimizations.
.. code-block:: none
- none: 3.420636
+ none: 3.231082
@@ -1083,7 +1083,7 @@ schedule.
.. code-block:: none
- blocking: 0.292891
+ blocking: 0.304476
@@ -1178,7 +1178,7 @@ already cache friendly from our previous optimizations.
.. code-block:: none
- vectorization: 0.331049
+ vectorization: 0.339453
@main = primfn(A_1: handle, B_1: handle, C_1: handle) -> ()
attr = {"from_legacy_te_schedule": True, "global_symbol": "main", "tir.noalias": True}
buffers = {A: Buffer(A_2: Pointer(float32), float32, [1024, 1024], []),
@@ -1251,7 +1251,7 @@ more cache friendly.
.. code-block:: none
- loop permutation: 0.117497
+ loop permutation: 0.119050
@main = primfn(A_1: handle, B_1: handle, C_1: handle) -> ()
attr = {"from_legacy_te_schedule": True, "global_symbol": "main", "tir.noalias": True}
buffers = {A: Buffer(A_2: Pointer(float32), float32, [1024, 1024], []),
@@ -1349,7 +1349,7 @@ optimized schedule.
.. code-block:: none
- array packing: 0.109955
+ array packing: 0.108013
@main = primfn(A_1: handle, B_1: handle, C_1: handle) -> ()
attr = {"from_legacy_te_schedule": True, "global_symbol": "main", "tir.noalias": True}
buffers = {A: Buffer(A_2: Pointer(float32), float32, [1024, 1024], []),
@@ -1441,7 +1441,7 @@ to `C` when all the block results are ready.
.. code-block:: none
- block caching: 0.110841
+ block caching: 0.111283
@main = primfn(A_1: handle, B_1: handle, C_1: handle) -> ()
attr = {"from_legacy_te_schedule": True, "global_symbol": "main", "tir.noalias": True}
buffers = {A: Buffer(A_2: Pointer(float32), float32, [1024, 1024], []),
@@ -1526,7 +1526,7 @@ of thread-level parallelization.
.. code-block:: none
- parallelization: 0.146717
+ parallelization: 0.146402
@main = primfn(A_1: handle, B_1: handle, C_1: handle) -> ()
attr = {"from_legacy_te_schedule": True, "global_symbol": "main", "tir.noalias": True}
buffers = {A: Buffer(A_2: Pointer(float32), float32, [1024, 1024], []),
@@ -1606,13 +1606,13 @@ working, we can compare the results.
.. code-block:: none
Operator Timing Performance
- none 3.4206357382 1.0
- blocking 0.2928914111 0.08562484681696178
- vectorization 0.3310493137 0.0967800546556308
- loop permutation 0.1174967353 0.03434938540454731
- array packing 0.10995484100000001 0.03214456300391114
- block caching 0.1108409057 0.0324035980979157
- parallelization 0.1467169169 0.04289171023431015
+ none 3.2310817368 1.0
+ blocking 0.3044761868 0.09423351422287052
+ vectorization 0.3394530913 0.10505865185452959
+ loop permutation 0.11905037219999999 0.0368453607484115
+ array packing 0.10801348880000002 0.03342951296149334
+ block caching 0.1112832214 0.0344414751668315
+ parallelization 0.1464021941 0.04531058203590784
@@ -1652,11 +1652,6 @@ operations with tunable parameters that allows you to automatically optimize
the computation for specific platforms.
-.. rst-class:: sphx-glr-timing
-
- **Total running time of the script:** ( 1 minutes 0.931 seconds)
-
-
.. _sphx_glr_download_tutorial_tensor_expr_get_started.py:
.. only:: html
diff --git a/docs/commit_hash b/docs/commit_hash
index 276326e2dc..cd3eea78aa 100644
--- a/docs/commit_hash
+++ b/docs/commit_hash
@@ -1 +1 @@
-1d9863470e0e97413d05b98f2852dc7de60611a0
+12311dcdefd7f2213ce5ce78f2c590444a04b32d
diff --git a/docs/how_to/compile_models/from_darknet.html b/docs/how_to/compile_models/from_darknet.html
index 0f1567c1f0..419224a129 100644
--- a/docs/how_to/compile_models/from_darknet.html
+++ b/docs/how_to/compile_models/from_darknet.html
@@ -585,7 +585,7 @@ class:['truck 0.9266'] left:471 top:83 right:689 bottom:169
class:['bicycle 0.9984'] left:111 top:113 right:577 bottom:447
</pre></div>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 9.325 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 10.274 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-compile-models-from-darknet-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/7716f96385bd5abb6e822041e285be54/from_darknet.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">from_darknet.py</span></code></a></p>
diff --git a/docs/how_to/compile_models/from_keras.html b/docs/how_to/compile_models/from_keras.html
index ef597fa16f..a02958940d 100644
--- a/docs/how_to/compile_models/from_keras.html
+++ b/docs/how_to/compile_models/from_keras.html
@@ -506,7 +506,7 @@ pip install -U tensorflow --user
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Relay top-1 id: 285, class name: Egyptian cat
1/1 [==============================] - ETA: 0s
-1/1 [==============================] - 1s 925ms/step
+1/1 [==============================] - 1s 956ms/step
Keras top-1 id: 285, class name: Egyptian cat
</pre></div>
</div>
diff --git a/docs/how_to/compile_models/from_mxnet.html b/docs/how_to/compile_models/from_mxnet.html
index 50c3cd07d5..f80dcd2a07 100644
--- a/docs/how_to/compile_models/from_mxnet.html
+++ b/docs/how_to/compile_models/from_mxnet.html
@@ -440,7 +440,7 @@ to download the full example code</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"x"</span><span class="p">,</span> <a href="https://docs.python.org/3/library/stdtypes.html#tuple" title="builtins.tuple" class="sphx-glr-backref-module-builtins sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span class="n">x</span><span class="o">.</span><span class="n">shape</span></a><span class="p">)</span>
</pre></div>
</div>
-<img src="../../_images/sphx_glr_from_mxnet_001.png" srcset="../../_images/sphx_glr_from_mxnet_001.png" alt="from mxnet" class = "sphx-glr-single-img"/><div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Downloading /workspace/.mxnet/models/resnet18_v1-a0666292.zipc507092a-21dc-4fa0-b0af-61eeba7ff005 from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/resnet18_v1-a0666292.zip...
+<img src="../../_images/sphx_glr_from_mxnet_001.png" srcset="../../_images/sphx_glr_from_mxnet_001.png" alt="from mxnet" class = "sphx-glr-single-img"/><div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Downloading /workspace/.mxnet/models/resnet18_v1-a0666292.zip02d83f1a-3f5a-43d2-b737-d5af5d3a6d59 from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/resnet18_v1-a0666292.zip...
x (1, 3, 224, 224)
</pre></div>
</div>
diff --git a/docs/how_to/compile_models/from_oneflow.html b/docs/how_to/compile_models/from_oneflow.html
index fa691bb736..00f944c5bd 100644
--- a/docs/how_to/compile_models/from_oneflow.html
+++ b/docs/how_to/compile_models/from_oneflow.html
@@ -448,12 +448,13 @@ Deprecated in NumPy 1.20; for more details and guidance: https://numpy.org/devdo
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Downloading: "https://oneflow-public.oss-cn-beijing.aliyuncs.com/model_zoo/flowvision/classification/ResNet/resnet18.zip" to /workspace/.oneflow/flowvision_cache/resnet18.zip
0%| | 0.00/41.5M [00:00<?, ?B/s]
- 19%|#9 | 7.99M/41.5M [00:00<00:00, 45.8MB/s]
- 39%|###8 | 16.0M/41.5M [00:00<00:00, 47.6MB/s]
- 58%|#####7 | 24.0M/41.5M [00:00<00:00, 49.1MB/s]
- 77%|#######7 | 32.0M/41.5M [00:00<00:00, 54.5MB/s]
- 96%|#########6| 40.0M/41.5M [00:00<00:00, 58.5MB/s]
-100%|##########| 41.5M/41.5M [00:00<00:00, 55.9MB/s]
+ 17%|#6 | 6.95M/41.5M [00:00<00:00, 72.8MB/s]
+ 33%|###3 | 13.9M/41.5M [00:00<00:00, 65.1MB/s]
+ 49%|####8 | 20.2M/41.5M [00:00<00:00, 37.5MB/s]
+ 59%|#####9 | 24.6M/41.5M [00:00<00:00, 36.8MB/s]
+ 77%|#######7 | 32.0M/41.5M [00:00<00:00, 46.1MB/s]
+ 92%|#########2| 38.3M/41.5M [00:00<00:00, 39.8MB/s]
+100%|##########| 41.5M/41.5M [00:01<00:00, 43.3MB/s]
</pre></div>
</div>
</div>
diff --git a/docs/how_to/compile_models/from_pytorch.html b/docs/how_to/compile_models/from_pytorch.html
index bf88733a39..d47ee039c9 100644
--- a/docs/how_to/compile_models/from_pytorch.html
+++ b/docs/how_to/compile_models/from_pytorch.html
@@ -431,10 +431,12 @@ be unstable.</p>
Downloading: "https://download.pytorch.org/models/resnet18-f37072fd.pth" to /workspace/.cache/torch/hub/checkpoints/resnet18-f37072fd.pth
0%| | 0.00/44.7M [00:00<?, ?B/s]
- 18%|#7 | 7.99M/44.7M [00:00<00:00, 60.4MB/s]
- 54%|#####3 | 24.0M/44.7M [00:00<00:00, 108MB/s]
- 78%|#######7 | 34.8M/44.7M [00:00<00:00, 103MB/s]
-100%|##########| 44.7M/44.7M [00:00<00:00, 94.8MB/s]
+ 18%|#7 | 7.99M/44.7M [00:00<00:00, 49.2MB/s]
+ 35%|###5 | 15.7M/44.7M [00:00<00:00, 52.0MB/s]
+ 54%|#####3 | 24.0M/44.7M [00:00<00:00, 58.1MB/s]
+ 72%|#######1 | 32.0M/44.7M [00:00<00:00, 60.4MB/s]
+ 94%|#########4| 42.1M/44.7M [00:00<00:00, 72.8MB/s]
+100%|##########| 44.7M/44.7M [00:00<00:00, 65.2MB/s]
</pre></div>
</div>
</div>
diff --git a/docs/how_to/compile_models/from_tensorflow.html b/docs/how_to/compile_models/from_tensorflow.html
index d520de2e4f..f110c96f67 100644
--- a/docs/how_to/compile_models/from_tensorflow.html
+++ b/docs/how_to/compile_models/from_tensorflow.html
@@ -645,7 +645,7 @@ banana (score = 0.00022)
desk (score = 0.00019)
</pre></div>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 11.613 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 12.870 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-compile-models-from-tensorflow-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/7f1d3d1b878694c201c614c807cdebc8/from_tensorflow.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">from_tensorflow.py</span></code></a></p>
diff --git a/docs/how_to/compile_models/sg_execution_times.html b/docs/how_to/compile_models/sg_execution_times.html
index 6bf20666da..7ef66dbf47 100644
--- a/docs/how_to/compile_models/sg_execution_times.html
+++ b/docs/how_to/compile_models/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-compile-models-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>05:41.464</strong> total execution time for <strong>how_to_compile_models</strong> files:</p>
+<p><strong>05:46.830</strong> total execution time for <strong>how_to_compile_models</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 81%" />
@@ -349,43 +349,43 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="from_tensorflow.html#sphx-glr-how-to-compile-models-from-tensorflow-py"><span class="std std-ref">Compile Tensorflow Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_tensorflow.py</span></code>)</p></td>
-<td><p>01:11.613</p></td>
+<td><p>01:12.870</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="from_darknet.html#sphx-glr-how-to-compile-models-from-darknet-py"><span class="std std-ref">Compile YOLO-V2 and YOLO-V3 in DarkNet Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_darknet.py</span></code>)</p></td>
-<td><p>01:09.325</p></td>
+<td><p>01:10.274</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="from_paddle.html#sphx-glr-how-to-compile-models-from-paddle-py"><span class="std std-ref">Compile PaddlePaddle Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_paddle.py</span></code>)</p></td>
-<td><p>00:46.150</p></td>
+<td><p>00:46.820</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="from_oneflow.html#sphx-glr-how-to-compile-models-from-oneflow-py"><span class="std std-ref">Compile OneFlow Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_oneflow.py</span></code>)</p></td>
-<td><p>00:32.015</p></td>
+<td><p>00:33.016</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="from_mxnet.html#sphx-glr-how-to-compile-models-from-mxnet-py"><span class="std std-ref">Compile MXNet Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_mxnet.py</span></code>)</p></td>
-<td><p>00:28.622</p></td>
+<td><p>00:28.659</p></td>
<td><p>0.0 MB</p></td>
</tr>
-<tr class="row-even"><td><p><a class="reference internal" href="from_coreml.html#sphx-glr-how-to-compile-models-from-coreml-py"><span class="std std-ref">Compile CoreML Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_coreml.py</span></code>)</p></td>
-<td><p>00:26.501</p></td>
+<tr class="row-even"><td><p><a class="reference internal" href="from_tflite.html#sphx-glr-how-to-compile-models-from-tflite-py"><span class="std std-ref">Compile TFLite Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_tflite.py</span></code>)</p></td>
+<td><p>00:26.100</p></td>
<td><p>0.0 MB</p></td>
</tr>
-<tr class="row-odd"><td><p><a class="reference internal" href="from_tflite.html#sphx-glr-how-to-compile-models-from-tflite-py"><span class="std std-ref">Compile TFLite Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_tflite.py</span></code>)</p></td>
-<td><p>00:25.450</p></td>
+<tr class="row-odd"><td><p><a class="reference internal" href="from_coreml.html#sphx-glr-how-to-compile-models-from-coreml-py"><span class="std std-ref">Compile CoreML Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_coreml.py</span></code>)</p></td>
+<td><p>00:26.060</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="from_pytorch.html#sphx-glr-how-to-compile-models-from-pytorch-py"><span class="std std-ref">Compile PyTorch Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_pytorch.py</span></code>)</p></td>
-<td><p>00:22.198</p></td>
+<td><p>00:22.672</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="from_keras.html#sphx-glr-how-to-compile-models-from-keras-py"><span class="std std-ref">Compile Keras Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_keras.py</span></code>)</p></td>
-<td><p>00:17.209</p></td>
+<td><p>00:17.903</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="from_onnx.html#sphx-glr-how-to-compile-models-from-onnx-py"><span class="std std-ref">Compile ONNX Models</span></a> (<code class="docutils literal notranslate"><span class="pre">from_onnx.py</span></code>)</p></td>
-<td><p>00:02.383</p></td>
+<td><p>00:02.456</p></td>
<td><p>0.0 MB</p></td>
</tr>
</tbody>
diff --git a/docs/how_to/deploy_models/deploy_model_on_adreno.html b/docs/how_to/deploy_models/deploy_model_on_adreno.html
index a0741baa3d..4a1da4b2bb 100644
--- a/docs/how_to/deploy_models/deploy_model_on_adreno.html
+++ b/docs/how_to/deploy_models/deploy_model_on_adreno.html
@@ -919,7 +919,7 @@ Top5 predictions:
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 2520.0950 2518.8338 2534.8240 2515.2016 5.4340
+ 2518.6538 2517.1939 2526.2019 2515.5964 2.9954
</pre></div>
</div>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-deploy-models-deploy-model-on-adreno-py">
diff --git a/docs/how_to/deploy_models/deploy_model_on_android.html b/docs/how_to/deploy_models/deploy_model_on_android.html
index 223cf4fb3a..803b527667 100644
--- a/docs/how_to/deploy_models/deploy_model_on_android.html
+++ b/docs/how_to/deploy_models/deploy_model_on_android.html
@@ -661,7 +661,7 @@ to the remote android device.</p>
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 15.6967 15.6694 15.9157 15.5146 0.1490
+ 16.0341 16.0132 16.2077 15.9043 0.0924
</pre></div>
</div>
</div>
diff --git a/docs/how_to/deploy_models/deploy_object_detection_pytorch.html b/docs/how_to/deploy_models/deploy_object_detection_pytorch.html
index 331b26a8ad..bfcf87e28a 100644
--- a/docs/how_to/deploy_models/deploy_object_detection_pytorch.html
+++ b/docs/how_to/deploy_models/deploy_object_detection_pytorch.html
@@ -453,27 +453,29 @@ be unstable.</p>
Downloading: "https://download.pytorch.org/models/maskrcnn_resnet50_fpn_coco-bf2d0c1e.pth" to /workspace/.cache/torch/hub/checkpoints/maskrcnn_resnet50_fpn_coco-bf2d0c1e.pth
0%| | 0.00/170M [00:00<?, ?B/s]
- 5%|4 | 8.12M/170M [00:00<00:02, 83.3MB/s]
- 11%|#1 | 19.2M/170M [00:00<00:01, 102MB/s]
- 17%|#7 | 29.0M/170M [00:00<00:01, 98.1MB/s]
- 23%|##2 | 38.4M/170M [00:00<00:01, 81.2MB/s]
- 27%|##7 | 46.4M/170M [00:00<00:01, 82.2MB/s]
- 33%|###2 | 56.0M/170M [00:00<00:01, 72.9MB/s]
- 38%|###7 | 64.0M/170M [00:00<00:01, 74.8MB/s]
- 42%|####2 | 72.0M/170M [00:00<00:01, 77.0MB/s]
- 47%|####7 | 80.0M/170M [00:01<00:01, 77.9MB/s]
- 52%|#####1 | 88.0M/170M [00:01<00:01, 78.1MB/s]
- 57%|#####6 | 96.0M/170M [00:01<00:00, 77.8MB/s]
- 61%|######1 | 104M/170M [00:01<00:00, 70.0MB/s]
- 66%|######5 | 112M/170M [00:01<00:00, 68.0MB/s]
- 71%|#######1 | 121M/170M [00:01<00:00, 75.3MB/s]
- 76%|#######5 | 129M/170M [00:01<00:00, 63.9MB/s]
- 80%|######## | 136M/170M [00:01<00:00, 65.3MB/s]
- 85%|########4 | 144M/170M [00:02<00:00, 68.5MB/s]
- 89%|########9 | 152M/170M [00:02<00:00, 61.3MB/s]
- 94%|#########4| 160M/170M [00:02<00:00, 60.7MB/s]
- 99%|#########8| 168M/170M [00:02<00:00, 66.6MB/s]
-100%|##########| 170M/170M [00:02<00:00, 72.6MB/s]
+ 5%|4 | 7.99M/170M [00:00<00:02, 75.0MB/s]
+ 9%|8 | 15.2M/170M [00:00<00:02, 73.0MB/s]
+ 13%|#3 | 22.1M/170M [00:00<00:02, 72.3MB/s]
+ 17%|#7 | 29.0M/170M [00:00<00:02, 52.1MB/s]
+ 20%|## | 34.5M/170M [00:01<00:05, 23.7MB/s]
+ 26%|##6 | 44.4M/170M [00:01<00:03, 35.8MB/s]
+ 31%|### | 52.1M/170M [00:01<00:02, 43.8MB/s]
+ 35%|###4 | 58.7M/170M [00:01<00:02, 42.7MB/s]
+ 38%|###7 | 64.3M/170M [00:01<00:02, 42.3MB/s]
+ 44%|####3 | 74.1M/170M [00:01<00:02, 49.7MB/s]
+ 48%|####8 | 82.1M/170M [00:01<00:01, 54.9MB/s]
+ 52%|#####1 | 88.0M/170M [00:02<00:01, 45.8MB/s]
+ 57%|#####6 | 96.1M/170M [00:02<00:01, 51.2MB/s]
+ 61%|######1 | 104M/170M [00:02<00:01, 48.0MB/s]
+ 67%|######7 | 114M/170M [00:02<00:00, 59.5MB/s]
+ 71%|####### | 120M/170M [00:02<00:00, 52.8MB/s]
+ 75%|#######5 | 128M/170M [00:02<00:00, 54.6MB/s]
+ 80%|######## | 136M/170M [00:02<00:00, 52.6MB/s]
+ 86%|########5 | 145M/170M [00:03<00:00, 62.2MB/s]
+ 89%|########9 | 152M/170M [00:03<00:00, 57.2MB/s]
+ 93%|#########2| 158M/170M [00:03<00:00, 57.9MB/s]
+ 96%|#########6| 164M/170M [00:03<00:00, 53.0MB/s]
+100%|##########| 170M/170M [00:03<00:00, 50.7MB/s]
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torch/nn/functional.py:3897: UserWarning: To copy construct from a tensor, it is recommended to use sourceTensor.clone().detach() or sourceTensor.clone().detach().requires_grad_(True), rather than torch.tensor(sourceTensor).
for i in range(dim)
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/torchvision/models/detection/anchor_utils.py:124: UserWarning: __floordiv__ is deprecated, and its behavior will change in a future version of pytorch. It currently rounds toward 0 (like the 'trunc' function NOT 'floor'). This results in incorrect rounding for negative values. To keep the current behavior, use torch.div(a, b, rounding_mode='trunc'), or for actual floor division, use torch.div(a, b, rounding_mode=& [...]
@@ -571,7 +573,7 @@ torchvision rcnn models.</p>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Get 9 valid boxes
</pre></div>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 3 minutes 12.782 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 3 minutes 16.741 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-deploy-models-deploy-object-detection-pytorch-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/7795da4b258c8feff986668b95ef57ad/deploy_object_detection_pytorch.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">deploy_object_detection_pytorch.py</span></code></a></p>
diff --git a/docs/how_to/deploy_models/deploy_prequantized.html b/docs/how_to/deploy_models/deploy_prequantized.html
index 73218502da..d7512e318c 100644
--- a/docs/how_to/deploy_models/deploy_prequantized.html
+++ b/docs/how_to/deploy_models/deploy_prequantized.html
@@ -497,8 +497,9 @@ training. Other models require a full post training calibration.</p>
Downloading: "https://download.pytorch.org/models/mobilenet_v2-b0353104.pth" to /workspace/.cache/torch/hub/checkpoints/mobilenet_v2-b0353104.pth
0%| | 0.00/13.6M [00:00<?, ?B/s]
- 59%|#####8 | 7.99M/13.6M [00:00<00:00, 58.8MB/s]
-100%|##########| 13.6M/13.6M [00:00<00:00, 75.0MB/s]
+ 59%|#####8 | 7.99M/13.6M [00:00<00:00, 48.3MB/s]
+ 93%|#########2| 12.6M/13.6M [00:00<00:00, 40.7MB/s]
+100%|##########| 13.6M/13.6M [00:00<00:00, 44.5MB/s]
</pre></div>
</div>
</div>
@@ -589,7 +590,7 @@ output values are identical out of 1000 outputs from mobilenet v2.</p>
</div>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 90.5552 90.3865 100.4027 90.0171 1.1241
+ 90.3728 90.2583 93.4944 90.0504 0.4752
</pre></div>
</div>
<div class="admonition note">
@@ -628,7 +629,7 @@ This includes support for the VNNI 8 bit dot product instruction (CascadeLake or
<div class="section" id="deploy-a-quantized-tflite-model">
<h2>Deploy a quantized TFLite Model<a class="headerlink" href="#deploy-a-quantized-tflite-model" title="Permalink to this headline">¶</a></h2>
<p>TODO</p>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 6.105 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 6.981 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-deploy-models-deploy-prequantized-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/fb8217c13f4351224c6cf3aacf1a87fc/deploy_prequantized.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">deploy_prequantized.py</span></code></a></p>
diff --git a/docs/how_to/deploy_models/deploy_prequantized_tflite.html b/docs/how_to/deploy_models/deploy_prequantized_tflite.html
index 9d92cc4ca9..64be4b739a 100644
--- a/docs/how_to/deploy_models/deploy_prequantized_tflite.html
+++ b/docs/how_to/deploy_models/deploy_prequantized_tflite.html
@@ -582,7 +582,7 @@ TFLite Top-5 labels: [387 102 386 341 349]
</div>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 121.0501 120.9877 124.7303 120.0469 0.5382
+ 123.0244 122.9984 125.3565 122.1152 0.5555
</pre></div>
</div>
<div class="admonition note">
@@ -610,7 +610,7 @@ network for ARM CPU</span></a>.</p></li>
</ul>
</div></blockquote>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 2 minutes 24.044 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 2 minutes 23.771 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-deploy-models-deploy-prequantized-tflite-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/56691c7a27d45da61d112276334640d3/deploy_prequantized_tflite.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">deploy_prequantized_tflite.py</span></code></a></p>
diff --git a/docs/how_to/deploy_models/deploy_quantized.html b/docs/how_to/deploy_models/deploy_quantized.html
index 028e606af0..428e3ba7d3 100644
--- a/docs/how_to/deploy_models/deploy_quantized.html
+++ b/docs/how_to/deploy_models/deploy_quantized.html
@@ -520,7 +520,7 @@ for calibration. But the accuracy might be impacted.</p>
DeprecationWarning,
</pre></div>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 39.176 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 34.524 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-deploy-models-deploy-quantized-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/7810ecf51bfc05f7d5e8a400ac3e815d/deploy_quantized.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">deploy_quantized.py</span></code></a></p>
diff --git a/docs/how_to/deploy_models/deploy_ssd_gluoncv.html b/docs/how_to/deploy_models/deploy_ssd_gluoncv.html
index c6090714fd..62702b4be1 100644
--- a/docs/how_to/deploy_models/deploy_ssd_gluoncv.html
+++ b/docs/how_to/deploy_models/deploy_ssd_gluoncv.html
@@ -462,24 +462,23 @@ to your device.</p>
Downloading /workspace/.mxnet/models/ssd_512_resnet50_v1_voc-9c8b225a.zip from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/ssd_512_resnet50_v1_voc-9c8b225a.zip...
0%| | 0/132723 [00:00<?, ?KB/s]
- 4%|4 | 5592/132723 [00:00<00:02, 55893.13KB/s]
- 10%|# | 13456/132723 [00:00<00:01, 69265.50KB/s]
- 15%|#5 | 20383/132723 [00:00<00:02, 45004.27KB/s]
- 21%|##1 | 28154/132723 [00:00<00:01, 54726.62KB/s]
- 27%|##7 | 35946/132723 [00:00<00:01, 61644.23KB/s]
- 33%|###3 | 43853/132723 [00:00<00:01, 66847.41KB/s]
- 39%|###9 | 51776/132723 [00:00<00:01, 70549.64KB/s]
- 45%|####4 | 59709/132723 [00:00<00:00, 73176.49KB/s]
- 51%|#####1 | 67814/132723 [00:01<00:00, 75534.23KB/s]
- 57%|#####7 | 75682/132723 [00:01<00:00, 76475.12KB/s]
- 63%|######3 | 83636/132723 [00:01<00:00, 77392.17KB/s]
- 69%|######8 | 91578/132723 [00:01<00:00, 77998.16KB/s]
- 75%|#######4 | 99468/132723 [00:01<00:00, 78267.40KB/s]
- 81%|######## | 107500/132723 [00:01<00:00, 78880.10KB/s]
- 87%|########6 | 115428/132723 [00:01<00:00, 78994.92KB/s]
- 93%|#########2| 123430/132723 [00:01<00:00, 79295.94KB/s]
- 99%|#########9| 131416/132723 [00:01<00:00, 79463.16KB/s]
-100%|##########| 132723/132723 [00:01<00:00, 72324.04KB/s]
+ 4%|4 | 5746/132723 [00:00<00:02, 57456.90KB/s]
+ 10%|# | 13768/132723 [00:00<00:01, 70841.54KB/s]
+ 17%|#6 | 21911/132723 [00:00<00:01, 75674.75KB/s]
+ 23%|##2 | 30035/132723 [00:00<00:01, 77870.10KB/s]
+ 29%|##8 | 38138/132723 [00:00<00:01, 79008.27KB/s]
+ 35%|###4 | 46294/132723 [00:00<00:01, 79874.56KB/s]
+ 41%|#### | 54394/132723 [00:00<00:00, 80239.38KB/s]
+ 47%|####7 | 62472/132723 [00:00<00:00, 80407.54KB/s]
+ 53%|#####3 | 70599/132723 [00:00<00:00, 80671.33KB/s]
+ 59%|#####9 | 78742/132723 [00:01<00:00, 80898.22KB/s]
+ 65%|######5 | 86832/132723 [00:01<00:00, 75357.26KB/s]
+ 72%|#######1 | 94953/132723 [00:01<00:00, 77049.49KB/s]
+ 78%|#######7 | 103065/132723 [00:01<00:00, 78238.09KB/s]
+ 84%|########3 | 111235/132723 [00:01<00:00, 79253.13KB/s]
+ 90%|########9 | 119361/132723 [00:01<00:00, 79844.53KB/s]
+ 96%|#########6| 127457/132723 [00:01<00:00, 80174.48KB/s]
+100%|##########| 132723/132723 [00:01<00:00, 78566.50KB/s]
</pre></div>
</div>
<p>Create TVM runtime and do inference
@@ -518,7 +517,7 @@ Downloading /workspace/.mxnet/models/ssd_512_resnet50_v1_voc-9c8b225a.zip from h
<span class="n">plt</span><span class="o">.</span><span class="n">show</span><span class="p">()</span>
</pre></div>
</div>
-<img src="../../_images/sphx_glr_deploy_ssd_gluoncv_001.png" srcset="../../_images/sphx_glr_deploy_ssd_gluoncv_001.png" alt="deploy ssd gluoncv" class = "sphx-glr-single-img"/><p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 3 minutes 5.609 seconds)</p>
+<img src="../../_images/sphx_glr_deploy_ssd_gluoncv_001.png" srcset="../../_images/sphx_glr_deploy_ssd_gluoncv_001.png" alt="deploy ssd gluoncv" class = "sphx-glr-single-img"/><p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 3 minutes 6.504 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-deploy-models-deploy-ssd-gluoncv-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/cccb17d28e5e8b2e94ea8cd5ec59f6ed/deploy_ssd_gluoncv.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">deploy_ssd_gluoncv.py</span></code></a></p>
diff --git a/docs/how_to/deploy_models/sg_execution_times.html b/docs/how_to/deploy_models/sg_execution_times.html
index edf6ade79d..5070699e8f 100644
--- a/docs/how_to/deploy_models/sg_execution_times.html
+++ b/docs/how_to/deploy_models/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-deploy-models-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>13:43.432</strong> total execution time for <strong>how_to_deploy_models</strong> files:</p>
+<p><strong>13:45.186</strong> total execution time for <strong>how_to_deploy_models</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 86%" />
@@ -349,39 +349,39 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="deploy_object_detection_pytorch.html#sphx-glr-how-to-deploy-models-deploy-object-detection-pytorch-py"><span class="std std-ref">Compile PyTorch Object Detection Models</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_object_detection_pytorch.py</span></code>)</p></td>
-<td><p>03:12.782</p></td>
+<td><p>03:16.741</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="deploy_ssd_gluoncv.html#sphx-glr-how-to-deploy-models-deploy-ssd-gluoncv-py"><span class="std std-ref">Deploy Single Shot Multibox Detector(SSD) model</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_ssd_gluoncv.py</span></code>)</p></td>
-<td><p>03:05.609</p></td>
+<td><p>03:06.504</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="deploy_prequantized_tflite.html#sphx-glr-how-to-deploy-models-deploy-prequantized-tflite-py"><span class="std std-ref">Deploy a Framework-prequantized Model with TVM - Part 3 (TFLite)</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_prequantized_tflite.py</span></code>)</p></td>
-<td><p>02:24.044</p></td>
+<td><p>02:23.771</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="deploy_quantized.html#sphx-glr-how-to-deploy-models-deploy-quantized-py"><span class="std std-ref">Deploy a Quantized Model on Cuda</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_quantized.py</span></code>)</p></td>
-<td><p>01:39.176</p></td>
+<td><p>01:34.524</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="deploy_prequantized.html#sphx-glr-how-to-deploy-models-deploy-prequantized-py"><span class="std std-ref">Deploy a Framework-prequantized Model with TVM</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_prequantized.py</span></code>)</p></td>
-<td><p>01:06.105</p></td>
+<td><p>01:06.981</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="deploy_model_on_adreno.html#sphx-glr-how-to-deploy-models-deploy-model-on-adreno-py"><span class="std std-ref">Deploy the Pretrained Model on Adreno</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_model_on_adreno.py</span></code>)</p></td>
-<td><p>00:51.040</p></td>
+<td><p>00:51.228</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="deploy_model_on_android.html#sphx-glr-how-to-deploy-models-deploy-model-on-android-py"><span class="std std-ref">Deploy the Pretrained Model on Android</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_model_on_android.py</span></code>)</p></td>
-<td><p>00:35.265</p></td>
+<td><p>00:35.657</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="deploy_model_on_nano.html#sphx-glr-how-to-deploy-models-deploy-model-on-nano-py"><span class="std std-ref">Deploy the Pretrained Model on Jetson Nano</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_model_on_nano.py</span></code>)</p></td>
-<td><p>00:24.929</p></td>
+<td><p>00:25.140</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="deploy_model_on_rasp.html#sphx-glr-how-to-deploy-models-deploy-model-on-rasp-py"><span class="std std-ref">Deploy the Pretrained Model on Raspberry Pi</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_model_on_rasp.py</span></code>)</p></td>
-<td><p>00:24.475</p></td>
+<td><p>00:24.632</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="deploy_sparse.html#sphx-glr-how-to-deploy-models-deploy-sparse-py"><span class="std std-ref">Deploy a Hugging Face Pruned Model on CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">deploy_sparse.py</span></code>)</p></td>
diff --git a/docs/how_to/extend_tvm/bring_your_own_datatypes.html b/docs/how_to/extend_tvm/bring_your_own_datatypes.html
index 15a5f7793c..2fabb9927d 100644
--- a/docs/how_to/extend_tvm/bring_your_own_datatypes.html
+++ b/docs/how_to/extend_tvm/bring_your_own_datatypes.html
@@ -621,7 +621,7 @@ In this alpha state of the Bring Your Own Datatypes framework, we have not imple
<span class="n">module</span><span class="p">,</span> <a href="https://docs.python.org/3/library/stdtypes.html#dict" title="builtins.dict" class="sphx-glr-backref-module-builtins sphx-glr-backref-type-py-class sphx-glr-backref-instance"><span class="n">params</span></a> <span class="o">=</span> <span class="n">get_mobilenet</span><span class="p">()</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Downloading /workspace/.mxnet/models/mobilenet0.25-9f83e440.zip519d4d3a-4bac-4731-9b50-35fb1592ae8e from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/mobilenet0.25-9f83e440.zip...
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Downloading /workspace/.mxnet/models/mobilenet0.25-9f83e440.zip9f705891-4420-4dc7-8e6e-41d68c0ffa5e from https://apache-mxnet.s3-accelerate.dualstack.amazonaws.com/gluon/models/mobilenet0.25-9f83e440.zip...
</pre></div>
</div>
<p>It’s easy to execute MobileNet with native TVM:</p>
diff --git a/docs/how_to/extend_tvm/sg_execution_times.html b/docs/how_to/extend_tvm/sg_execution_times.html
index ed1b8e99ce..5e50f5f501 100644
--- a/docs/how_to/extend_tvm/sg_execution_times.html
+++ b/docs/how_to/extend_tvm/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-extend-tvm-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>00:47.168</strong> total execution time for <strong>how_to_extend_tvm</strong> files:</p>
+<p><strong>00:47.320</strong> total execution time for <strong>how_to_extend_tvm</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 84%" />
@@ -349,15 +349,15 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="bring_your_own_datatypes.html#sphx-glr-how-to-extend-tvm-bring-your-own-datatypes-py"><span class="std std-ref">Bring Your Own Datatypes to TVM</span></a> (<code class="docutils literal notranslate"><span class="pre">bring_your_own_datatypes.py</span></code>)</p></td>
-<td><p>00:43.742</p></td>
+<td><p>00:43.874</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="use_pass_instrument.html#sphx-glr-how-to-extend-tvm-use-pass-instrument-py"><span class="std std-ref">How to Use TVM Pass Instrument</span></a> (<code class="docutils literal notranslate"><span class="pre">use_pass_instrument.py</span></code>)</p></td>
-<td><p>00:02.398</p></td>
+<td><p>00:02.412</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="use_pass_infra.html#sphx-glr-how-to-extend-tvm-use-pass-infra-py"><span class="std std-ref">How to Use TVM Pass Infra</span></a> (<code class="docutils literal notranslate"><span class="pre">use_pass_infra.py</span></code>)</p></td>
-<td><p>00:01.020</p></td>
+<td><p>00:01.026</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="low_level_custom_pass.html#sphx-glr-how-to-extend-tvm-low-level-custom-pass-py"><span class="std std-ref">Writing a Customized Pass</span></a> (<code class="docutils literal notranslate"><span class="pre">low_level_custom_pass.py</span></code>)</p></td>
diff --git a/docs/how_to/extend_tvm/use_pass_instrument.html b/docs/how_to/extend_tvm/use_pass_instrument.html
index 556b18c435..0c60fd9f17 100644
--- a/docs/how_to/extend_tvm/use_pass_instrument.html
+++ b/docs/how_to/extend_tvm/use_pass_instrument.html
@@ -525,10 +525,10 @@ profile the execution time of each passes.</p>
</pre></div>
</div>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Printing results of timing profile...
-InferType: 7117us [7117us] (46.52%; 46.52%)
-FoldScaleAxis: 8182us [7us] (53.48%; 53.48%)
- FoldConstant: 8176us [1664us] (53.44%; 99.92%)
- InferType: 6511us [6511us] (42.56%; 79.64%)
+InferType: 7125us [7125us] (46.15%; 46.15%)
+FoldScaleAxis: 8313us [6us] (53.85%; 53.85%)
+ FoldConstant: 8307us [1681us] (53.81%; 99.93%)
+ InferType: 6626us [6626us] (42.92%; 79.76%)
</pre></div>
</div>
</div>
@@ -550,10 +550,10 @@ Refer to following sections and <a class="reference internal" href="../../refere
</pre></div>
</div>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Printing results of timing profile...
-InferType: 6612us [6612us] (45.22%; 45.22%)
-FoldScaleAxis: 8010us [5us] (54.78%; 54.78%)
- FoldConstant: 8005us [1642us] (54.75%; 99.94%)
- InferType: 6363us [6363us] (43.52%; 79.49%)
+InferType: 6724us [6724us] (44.23%; 44.23%)
+FoldScaleAxis: 8479us [6us] (55.77%; 55.77%)
+ FoldConstant: 8473us [1715us] (55.73%; 99.93%)
+ InferType: 6758us [6758us] (44.45%; 79.76%)
</pre></div>
</div>
<p>Register empty list to clear existing instruments.</p>
diff --git a/docs/how_to/optimize_operators/opt_conv_cuda.html b/docs/how_to/optimize_operators/opt_conv_cuda.html
index cd89d35b68..02ada66e69 100644
--- a/docs/how_to/optimize_operators/opt_conv_cuda.html
+++ b/docs/how_to/optimize_operators/opt_conv_cuda.html
@@ -577,7 +577,7 @@ latency of convolution.</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"Convolution: </span><span class="si">%f</span><span class="s2"> ms"</span> <span class="o">%</span> <span class="p">(</span><span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">w</span><span class="p">,</span> <span class="n">b</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span> <span class="o">*</span> <span cl [...]
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Convolution: 54.214591 ms
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Convolution: 46.378784 ms
</pre></div>
</div>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-optimize-operators-opt-conv-cuda-py">
diff --git a/docs/how_to/optimize_operators/opt_conv_tensorcore.html b/docs/how_to/optimize_operators/opt_conv_tensorcore.html
index b80a545090..418966437f 100644
--- a/docs/how_to/optimize_operators/opt_conv_tensorcore.html
+++ b/docs/how_to/optimize_operators/opt_conv_tensorcore.html
@@ -914,7 +914,7 @@ be able to run on our build server</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"conv2d with tensor core: </span><span class="si">%f</span><span class="s2"> ms"</span> <span class="o">%</span> <span class="p">(</span><span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">w</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span> <span class="o">* [...]
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>conv2d with tensor core: 13.313641 ms
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>conv2d with tensor core: 13.351321 ms
</pre></div>
</div>
</div>
diff --git a/docs/how_to/optimize_operators/opt_gemm.html b/docs/how_to/optimize_operators/opt_gemm.html
index 5c7cc300bb..b3e73ac4ea 100644
--- a/docs/how_to/optimize_operators/opt_gemm.html
+++ b/docs/how_to/optimize_operators/opt_gemm.html
@@ -474,8 +474,8 @@ Then we write a baseline implementation, the simplest way to write a matrix mult
<span class="nb">print</span><span class="p">(</span><span class="s2">"Baseline: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">b</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Numpy running time: 0.018498
-Baseline: 3.376149
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Numpy running time: 0.018197
+Baseline: 3.276830
</pre></div>
</div>
<p>In TVM, we can always inspect lower level IR to debug or optimize our schedule.
@@ -534,7 +534,7 @@ fill 32 * 32 * sizeof(float) which is 4KB in the cache whose total size is 32KB
<span class="nb">print</span><span class="p">(</span><span class="s2">"Opt1: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">b</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt1: 0.291363
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt1: 0.303195
</pre></div>
</div>
<p>Here is the generated IR after blocking.</p>
@@ -600,7 +600,7 @@ vastly.</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"Opt2: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">b</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt2: 0.329749
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt2: 0.336533
</pre></div>
</div>
<p>Here is the generated IR after vectorization.</p>
@@ -660,7 +660,7 @@ the access pattern for A matrix is more cache friendly.</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"Opt3: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">b</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt3: 0.115922
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt3: 0.120270
</pre></div>
</div>
<p>Here is the generated IR after loop permutation.</p>
@@ -742,7 +742,7 @@ flattening.</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"Opt4: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">b</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt4: 0.109779
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt4: 0.110110
</pre></div>
</div>
<p>Here is the generated IR after array packing.</p>
@@ -827,7 +827,7 @@ write to C when all the block results are ready.</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"Opt5: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">evaluator</span><span class="p">(</span><span class="n">a</span><span class="p">,</span> <span class="n">b</span><span class="p">,</span> <span class="n">c</span><span class="p">)</span><span class="o">.</span><span class="n">mean</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt5: 0.111204
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt5: 0.110684
</pre></div>
</div>
<p>Here is the generated IR after blocking.</p>
@@ -916,7 +916,7 @@ write to C when all the block results are ready.</p>
<span class="nb">print</span><span class="p">(</span><span class="s2">"Opt6: </span><span class="si">%f</span><span class="s2">"</span> <span class="o">%</span> <span class="n">opt6_time</span><span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt6: 0.146807
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Opt6: 0.147054
</pre></div>
</div>
<p>Here is the generated IR after parallelization.</p>
diff --git a/docs/how_to/optimize_operators/sg_execution_times.html b/docs/how_to/optimize_operators/sg_execution_times.html
index e71589dc6e..e046a8cebf 100644
--- a/docs/how_to/optimize_operators/sg_execution_times.html
+++ b/docs/how_to/optimize_operators/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-optimize-operators-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>00:34.560</strong> total execution time for <strong>how_to_optimize_operators</strong> files:</p>
+<p><strong>00:34.533</strong> total execution time for <strong>how_to_optimize_operators</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 83%" />
@@ -349,15 +349,15 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="opt_gemm.html#sphx-glr-how-to-optimize-operators-opt-gemm-py"><span class="std std-ref">How to optimize GEMM on CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">opt_gemm.py</span></code>)</p></td>
-<td><p>00:32.022</p></td>
+<td><p>00:31.944</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="opt_conv_tensorcore.html#sphx-glr-how-to-optimize-operators-opt-conv-tensorcore-py"><span class="std std-ref">How to optimize convolution using TensorCores</span></a> (<code class="docutils literal notranslate"><span class="pre">opt_conv_tensorcore.py</span></code>)</p></td>
-<td><p>00:01.495</p></td>
+<td><p>00:01.519</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="opt_conv_cuda.html#sphx-glr-how-to-optimize-operators-opt-conv-cuda-py"><span class="std std-ref">How to optimize convolution on GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">opt_conv_cuda.py</span></code>)</p></td>
-<td><p>00:01.044</p></td>
+<td><p>00:01.069</p></td>
<td><p>0.0 MB</p></td>
</tr>
</tbody>
diff --git a/docs/how_to/tune_with_autoscheduler/sg_execution_times.html b/docs/how_to/tune_with_autoscheduler/sg_execution_times.html
index d8360acdd0..cf31fd0fbb 100644
--- a/docs/how_to/tune_with_autoscheduler/sg_execution_times.html
+++ b/docs/how_to/tune_with_autoscheduler/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-tune-with-autoscheduler-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>09:05.595</strong> total execution time for <strong>how_to_tune_with_autoscheduler</strong> files:</p>
+<p><strong>08:51.610</strong> total execution time for <strong>how_to_tune_with_autoscheduler</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 85%" />
@@ -349,27 +349,27 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="tune_conv2d_layer_cuda.html#sphx-glr-how-to-tune-with-autoscheduler-tune-conv2d-layer-cuda-py"><span class="std std-ref">Auto-scheduling a Convolution Layer for GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_conv2d_layer_cuda.py</span></code>)</p></td>
-<td><p>05:42.541</p></td>
+<td><p>05:27.158</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="tune_network_x86.html#sphx-glr-how-to-tune-with-autoscheduler-tune-network-x86-py"><span class="std std-ref">Auto-scheduling a Neural Network for x86 CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_network_x86.py</span></code>)</p></td>
-<td><p>01:31.335</p></td>
+<td><p>01:31.258</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="tune_network_cuda.html#sphx-glr-how-to-tune-with-autoscheduler-tune-network-cuda-py"><span class="std std-ref">Auto-scheduling a Neural Network for NVIDIA GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_network_cuda.py</span></code>)</p></td>
-<td><p>01:01.296</p></td>
+<td><p>01:01.762</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="tune_sparse_x86.html#sphx-glr-how-to-tune-with-autoscheduler-tune-sparse-x86-py"><span class="std std-ref">Auto-scheduling Sparse Matrix Multiplication on CPU with Custom Sketch Rule</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_sparse_x86.py</span></code>)</p></td>
-<td><p>00:27.353</p></td>
+<td><p>00:28.295</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="tune_network_arm.html#sphx-glr-how-to-tune-with-autoscheduler-tune-network-arm-py"><span class="std std-ref">Auto-scheduling a Neural Network for ARM CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_network_arm.py</span></code>)</p></td>
-<td><p>00:12.000</p></td>
+<td><p>00:11.940</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="tune_network_mali.html#sphx-glr-how-to-tune-with-autoscheduler-tune-network-mali-py"><span class="std std-ref">Auto-scheduling a Neural Network for mali GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_network_mali.py</span></code>)</p></td>
-<td><p>00:11.071</p></td>
+<td><p>00:11.198</p></td>
<td><p>0.0 MB</p></td>
</tr>
</tbody>
diff --git a/docs/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.html b/docs/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.html
index 5dbd0f9524..9f3442659f 100644
--- a/docs/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.html
+++ b/docs/how_to/tune_with_autoscheduler/tune_conv2d_layer_cuda.html
@@ -503,1282 +503,125 @@ cooperative fetching, unrolling and operator fusion.</p>
bias: Buffer(bias_2: Pointer(float32), float32, [1, 512, 1, 1], []),
compute: Buffer(compute_2: Pointer(float32), float32, [1, 512, 7, 7], [])}
buffer_map = {data_1: data, kernel_1: kernel, bias_1: bias, compute_1: compute} {
- attr [IterVar(blockIdx.x: int32, (nullptr), "ThreadIndex", "blockIdx.x")] "thread_extent" = 16;
- allocate(conv2d_nchw: Pointer(local float32), float32, [28]), storage_scope = local;
- allocate(pad_temp.shared: Pointer(shared float32), float32, [1296]), storage_scope = shared;
- allocate(kernel.shared: Pointer(shared float32), float32, [4608]), storage_scope = shared;
- attr [IterVar(threadIdx.x: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56 {
- conv2d_nchw_1: Buffer(conv2d_nchw, float32, [196], [], scope="local", align=32)[0] = 0f32
- conv2d_nchw_1[14] = 0f32
+ attr [IterVar(blockIdx.x: int32, (nullptr), "ThreadIndex", "blockIdx.x")] "thread_extent" = 28;
+ allocate(conv2d_nchw: Pointer(local float32), float32, [4]), storage_scope = local;
+ allocate(pad_temp.shared: Pointer(shared float32), float32, [144]), storage_scope = shared;
+ allocate(kernel.shared: Pointer(shared float32), float32, [6144]), storage_scope = shared;
+ attr [IterVar(threadIdx.x: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224 {
+ conv2d_nchw_1: Buffer(conv2d_nchw, float32, [1], [], scope="local", align=4)[0] = 0f32
conv2d_nchw_1[1] = 0f32
- conv2d_nchw_1[15] = 0f32
conv2d_nchw_1[2] = 0f32
- conv2d_nchw_1[16] = 0f32
conv2d_nchw_1[3] = 0f32
- conv2d_nchw_1[17] = 0f32
- conv2d_nchw_1[4] = 0f32
- conv2d_nchw_1[18] = 0f32
- conv2d_nchw_1[5] = 0f32
- conv2d_nchw_1[19] = 0f32
- conv2d_nchw_1[6] = 0f32
- conv2d_nchw_1[20] = 0f32
- conv2d_nchw_1[7] = 0f32
- conv2d_nchw_1[21] = 0f32
- conv2d_nchw_1[8] = 0f32
- conv2d_nchw_1[22] = 0f32
- conv2d_nchw_1[9] = 0f32
- conv2d_nchw_1[23] = 0f32
- conv2d_nchw_1[10] = 0f32
- conv2d_nchw_1[24] = 0f32
- conv2d_nchw_1[11] = 0f32
- conv2d_nchw_1[25] = 0f32
- conv2d_nchw_1[12] = 0f32
- conv2d_nchw_1[26] = 0f32
- conv2d_nchw_1[13] = 0f32
- conv2d_nchw_1[27] = 0f32
for (rc.outer.outer: int32, 0, 32) {
- let cse_var_2: int32 = (rc.outer.outer*784)
- let cse_var_1: int32 = (rc.outer.outer*144)
- {
- attr [IterVar(threadIdx.x_1: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1: Buffer(pad_temp.shared, float32, [1296], [], scope="shared")[threadIdx.x_1] = @tir.if_then_else((((9 <= threadIdx.x_1) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3: Buffer(data_2, float32, [25088], [])[(((cse_var_2 + (floordiv(threadIdx.x_1, 9)*7)) + floormod(threadIdx.x_1, 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 56)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 56), 81)) && (floormod((threadIdx.x_1 + 56), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 2), 9))) && (floormod((threadIdx.x_1 + 2), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 56), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 56), 81), 9)*7)) + floormod((threadIdx.x_1 + 2), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 112)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 31), 81)) && (floormod((threadIdx.x_1 + 31), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 4), 9))) && (floormod((threadIdx.x_1 + 4), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 112), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 31), 81), 9)*7)) + floormod((threadIdx.x_1 + 4), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 168)] = @tir.if_then_else((((9 <= floormod((threadIdx.x_1 + 6), 81)) && (1 <= floormod((threadIdx.x_1 + 6), 9))) && (floormod((threadIdx.x_1 + 6), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 168), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 6), 81), 9)*7)) + floormod((threadIdx.x_1 + 6), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 224)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 62), 81)) && (floormod((threadIdx.x_1 + 62), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 8), 9))) && (floormod((threadIdx.x_1 + 8), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 224), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 62), 81), 9)*7)) + floormod((threadIdx.x_1 + 8), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 280)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 37), 81)) && (floormod((threadIdx.x_1 + 37), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 1), 9))) && (floormod((threadIdx.x_1 + 1), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 280), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 37), 81), 9)*7)) + floormod((threadIdx.x_1 + 1), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 336)] = @tir.if_then_else(((1 <= floormod((threadIdx.x_1 + 3), 9)) && (floormod((threadIdx.x_1 + 3), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 336), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 12), 81), 9)*7)) + floormod((threadIdx.x_1 + 3), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 392)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 68), 81)) && (floormod((threadIdx.x_1 + 68), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 5), 9))) && (floormod((threadIdx.x_1 + 5), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 392), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 68), 81), 9)*7)) + floormod((threadIdx.x_1 + 5), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 448)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 43), 81)) && (floormod((threadIdx.x_1 + 43), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 7), 9))) && (floormod((threadIdx.x_1 + 7), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 448), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 43), 81), 9)*7)) + floormod((threadIdx.x_1 + 7), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 504)] = @tir.if_then_else((((threadIdx.x_1 < 54) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 504), 81)*49)) + ((floordiv(threadIdx.x_1, 9) + 2)*7)) + floormod(threadIdx.x_1, 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 560)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 74), 81)) && (floormod((threadIdx.x_1 + 74), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 2), 9))) && (floormod((threadIdx.x_1 + 2), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 560), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 74), 81), 9)*7)) + floormod((threadIdx.x_1 + 2), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 616)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 49), 81)) && (floormod((threadIdx.x_1 + 49), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 4), 9))) && (floormod((threadIdx.x_1 + 4), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 616), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 49), 81), 9)*7)) + floormod((threadIdx.x_1 + 4), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 672)] = @tir.if_then_else((((threadIdx.x_1 < 48) && (1 <= floormod((threadIdx.x_1 + 6), 9))) && (floormod((threadIdx.x_1 + 6), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 672), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 24), 81), 9)*7)) + floormod((threadIdx.x_1 + 6), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 728)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 80), 81)) && (floormod((threadIdx.x_1 + 80), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 8), 9))) && (floormod((threadIdx.x_1 + 8), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 728), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 80), 81), 9)*7)) + floormod((threadIdx.x_1 + 8), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 784)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 55), 81)) && (floormod((threadIdx.x_1 + 55), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 1), 9))) && (floormod((threadIdx.x_1 + 1), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 784), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 55), 81), 9)*7)) + floormod((threadIdx.x_1 + 1), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 840)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 30), 81)) && (floormod((threadIdx.x_1 + 30), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 3), 9))) && (floormod((threadIdx.x_1 + 3), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 840), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 30), 81), 9)*7)) + floormod((threadIdx.x_1 + 3), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 896)] = @tir.if_then_else((((9 <= floormod((threadIdx.x_1 + 5), 81)) && (1 <= floormod((threadIdx.x_1 + 5), 9))) && (floormod((threadIdx.x_1 + 5), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 896), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 5), 81), 9)*7)) + floormod((threadIdx.x_1 + 5), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 952)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 61), 81)) && (floormod((threadIdx.x_1 + 61), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 7), 9))) && (floormod((threadIdx.x_1 + 7), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 952), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 61), 81), 9)*7)) + floormod((threadIdx.x_1 + 7), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1008)] = @tir.if_then_else(((((1 <= floormod((floordiv(threadIdx.x_1, 9) + 4), 9)) && (floormod((threadIdx.x_1 + 36), 81) < 72)) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1008), 81)*49)) + (floormod((floordiv(threadIdx.x_1, 9) + 4), 9)*7)) + floormod(threadIdx.x_1, 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1064)] = @tir.if_then_else(((1 <= floormod((threadIdx.x_1 + 2), 9)) && (floormod((threadIdx.x_1 + 2), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1064), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 11), 81), 9)*7)) + floormod((threadIdx.x_1 + 2), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1120)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 67), 81)) && (floormod((threadIdx.x_1 + 67), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 4), 9))) && (floormod((threadIdx.x_1 + 4), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1120), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 67), 81), 9)*7)) + floormod((threadIdx.x_1 + 4), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1176)] = @tir.if_then_else(((((9 <= floormod((threadIdx.x_1 + 42), 81)) && (floormod((threadIdx.x_1 + 42), 81) < 72)) && (1 <= floormod((threadIdx.x_1 + 6), 9))) && (floormod((threadIdx.x_1 + 6), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1176), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 42), 81), 9)*7)) + floormod((threadIdx.x_1 + 6), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- pad_temp.shared_1[(threadIdx.x_1 + 1232)] = @tir.if_then_else((((threadIdx.x_1 < 55) && (1 <= floormod((threadIdx.x_1 + 8), 9))) && (floormod((threadIdx.x_1 + 8), 9) < 8)), data_3[((((cse_var_2 + (floordiv((threadIdx.x_1 + 1232), 81)*49)) + (floordiv(floormod((threadIdx.x_1 + 17), 81), 9)*7)) + floormod((threadIdx.x_1 + 8), 9)) - 8)], 0f32, dtype=float32)
- attr [IterVar(threadIdx.x_1, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- if @tir.likely((threadIdx.x_1 < 8), dtype=bool) {
- pad_temp.shared_1[(threadIdx.x_1 + 1288)] = 0f32
- }
- attr [IterVar(threadIdx.x_2: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1: Buffer(kernel.shared, float32, [4608], [], scope="shared")[threadIdx.x_2] = kernel_3: Buffer(kernel_2, float32, [2359296], [])[(((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 56)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + (floordiv((threadIdx.x_2 + 56), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 112)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 112), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 168)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 168), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 224)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 224), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 280)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 280), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 336)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 336), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 392)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 392), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 448)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 448), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 504)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 504), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 560)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 560), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 616)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 616), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 672)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 672), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 728)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 728), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 784)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 784), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 840)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 840), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 896)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 896), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 952)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 952), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1008)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 32256)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1064)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1064), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1120)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1120), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1176)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1176), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1232)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1232), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1288)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1288), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1344)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1344), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1400)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1400), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1456)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1456), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1512)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1512), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1568)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1568), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1624)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1624), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1680)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1680), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1736)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1736), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1792)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1792), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1848)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1848), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1904)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1904), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 1960)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 1960), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2016)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 64512)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2072)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2072), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2128)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2128), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2184)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2184), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2240)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2240), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2296)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2296), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2352)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2352), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2408)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2408), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2464)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2464), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2520)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2520), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2576)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2576), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2632)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2632), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2688)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2688), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2744)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2744), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2800)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2800), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2856)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2856), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2912)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2912), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 2968)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 2968), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3024)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 96768)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3080)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3080), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3136)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3136), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3192)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3192), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3248)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3248), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3304)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3304), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3360)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3360), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3416)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3416), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3472)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3472), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3528)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3528), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3584)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3584), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3640)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3640), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 40), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3696)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3696), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 32), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3752)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3752), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 8), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3808)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3808), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 64), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3864)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3864), 144)*4608)) + cse_var_1) + (floormod((floordiv(threadIdx.x_2, 3) + 40), 48)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3920)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3920), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 32), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 3976)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 3976), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 88), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4032)] = kernel_3[((((blockIdx.x*147456) + cse_var_1) + threadIdx.x_2) + 129024)]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4088)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4088), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 56), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4144)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4144), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 112), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4200)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4200), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 8)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4256)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4256), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 80), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4312)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4312), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 136), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4368)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4368), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 16)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4424)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4424), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 104), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4480)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4480), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 16), 144), 3)*3)) + floormod((threadIdx.x_2 + 1), 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- kernel.shared_1[(threadIdx.x_2 + 4536)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4536), 144)*4608)) + cse_var_1) + ((floordiv(threadIdx.x_2, 3) + 24)*3)) + floormod(threadIdx.x_2, 3))]
- attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 56;
- if @tir.likely((threadIdx.x_2 < 16), dtype=bool) {
- kernel.shared_1[(threadIdx.x_2 + 4592)] = kernel_3[(((((blockIdx.x*147456) + (floordiv((threadIdx.x_2 + 4592), 144)*4608)) + cse_var_1) + (floordiv(floormod((threadIdx.x_2 + 128), 144), 3)*3)) + floormod((threadIdx.x_2 + 2), 3))]
- }
- for (rc.outer.inner: int32, 0, 4) {
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36))]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2304)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 1)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2305)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2306)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 3)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2307)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 4)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2308)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 5)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2309)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 6)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2310)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 7)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2311)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 8)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2312)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 9)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2313)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 10)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2314)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 11)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2315)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 12)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2316)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 13)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2317)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 14)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2318)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 15)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2319)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 16)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2320)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 17)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2321)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 18)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2322)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 19)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2323)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 20)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2324)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 21)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2325)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 22)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2326)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 23)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2327)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 24)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2328)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 25)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2329)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 26)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2330)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 27)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2331)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 28)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2332)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 29)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2333)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 30)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2334)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 31)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2335)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 32)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2336)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 33)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2337)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 34)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2338)]))
- conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[14] = (conv2d_nchw_1[14] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[15] = (conv2d_nchw_1[15] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[16] = (conv2d_nchw_1[16] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[17] = (conv2d_nchw_1[17] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[4] = (conv2d_nchw_1[4] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[18] = (conv2d_nchw_1[18] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[5] = (conv2d_nchw_1[5] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[19] = (conv2d_nchw_1[19] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[6] = (conv2d_nchw_1[6] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 35)]))
- conv2d_nchw_1[20] = (conv2d_nchw_1[20] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2339)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 144)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2448)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 1)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 145)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2449)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 2)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 3)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 4)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 5)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 6)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 7)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 146)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 8)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2450)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 9)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 147)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2451)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 10)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 148)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2452)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 11)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 12)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 13)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 14)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 15)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 16)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 149)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 17)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2453)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 18)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 150)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2454)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 19)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 151)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2455)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 20)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 21)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 22)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 23)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 24)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 25)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 152)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 26)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2456)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 81)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 153)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2457)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 82)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 154)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2458)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 83)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 84)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 85)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 86)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 87)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 88)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 155)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 89)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2459)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 90)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 156)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2460)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 91)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 157)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2461)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 92)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 93)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 94)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 95)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 96)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 97)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 158)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 98)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2462)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 99)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 159)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2463)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 100)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 160)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2464)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 101)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 102)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 103)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 104)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 105)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 106)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 161)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 107)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2465)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 162)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 162)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2466)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 163)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 163)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2467)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 164)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 165)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 166)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 167)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 168)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 169)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 164)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 170)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2468)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 171)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 165)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2469)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 172)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 166)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2470)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 173)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 174)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 175)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 176)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 177)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 178)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 167)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 179)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2471)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 180)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 168)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2472)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 181)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 169)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2473)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 182)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 183)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 184)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 185)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 186)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 187)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 170)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 188)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2474)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 243)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 171)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2475)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 244)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 172)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2476)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 245)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 246)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 247)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 248)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 249)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 250)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 173)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 251)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2477)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 252)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 174)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2478)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 253)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 175)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2479)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 254)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 255)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 256)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 257)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 258)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 259)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 176)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 260)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2480)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 261)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 177)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2481)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 262)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 178)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2482)]))
- conv2d_nchw_1[7] = (conv2d_nchw_1[7] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[21] = (conv2d_nchw_1[21] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 263)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[8] = (conv2d_nchw_1[8] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[22] = (conv2d_nchw_1[22] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 264)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[9] = (conv2d_nchw_1[9] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[23] = (conv2d_nchw_1[23] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 265)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[10] = (conv2d_nchw_1[10] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[24] = (conv2d_nchw_1[24] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 266)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[11] = (conv2d_nchw_1[11] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[25] = (conv2d_nchw_1[25] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 267)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[12] = (conv2d_nchw_1[12] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[26] = (conv2d_nchw_1[26] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 268)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- conv2d_nchw_1[13] = (conv2d_nchw_1[13] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 179)]))
- conv2d_nchw_1[27] = (conv2d_nchw_1[27] + (pad_temp.shared_1[(((rc.outer.inner*324) + (floormod(threadIdx.x, 7)*9)) + 269)]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*288) + (rc.outer.inner*36)) + 2483)]))
- }
- }
- }
- for (i1.inner: int32, 0, 2) {
- for (i3.inner: int32, 0, 7) {
- let cse_var_3: int32 = ((i1.inner*7) + i3.inner)
+ for (ry.outer.outer: int32, 0, 3) {
+ let cse_var_2: int32 = (rc.outer.outer*144)
+ let cse_var_1: int32 = (ry.outer.outer*3)
{
- compute_3: Buffer(compute_2, float32, [25088], [])[(((((blockIdx.x*1568) + (floordiv(threadIdx.x, 7)*98)) + (i1.inner*49)) + (floormod(threadIdx.x, 7)*7)) + i3.inner)] = max((conv2d_nchw_1[cse_var_3] + bias_3: Buffer(bias_2, float32, [512], [])[(((blockIdx.x*32) + (floordiv(threadIdx.x, 7)*2)) + i1.inner)]), 0f32)
- compute_3[((((((blockIdx.x*1568) + (floordiv(threadIdx.x, 7)*98)) + (i1.inner*49)) + (floormod(threadIdx.x, 7)*7)) + i3.inner) + 784)] = max((conv2d_nchw_1[(cse_var_3 + 14)] + bias_3[((((blockIdx.x*32) + (floordiv(threadIdx.x, 7)*2)) + i1.inner) + 16)]), 0f32)
+ attr [IterVar(threadIdx.x_1: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ if @tir.likely((threadIdx.x_1 < 144), dtype=bool) {
+ pad_temp.shared_1: Buffer(pad_temp.shared, float32, [144], [], scope="shared")[threadIdx.x_1] = @tir.if_then_else(((((1 <= (ry.outer.outer + floormod(blockIdx.x, 7))) && ((ry.outer.outer + floormod(blockIdx.x, 7)) < 8)) && (1 <= floormod(threadIdx.x_1, 9))) && (floormod(threadIdx.x_1, 9) < 8)), data_3: Buffer(data_2, float32, [25088], [])[((((((rc.outer.outer*784) + (floordiv(threadIdx.x_1, 9)*49)) + (ry.outer.outer*7)) + (floormo [...]
+ }
+ attr [IterVar(threadIdx.x_2: int32, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1: Buffer(kernel.shared, float32, [6144], [], scope="shared")[threadIdx.x_2] = kernel_3: Buffer(kernel_2, float32, [2359296], [])[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 224)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 224), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 448)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 448), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 672)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 64512)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 896)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 896), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1120)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 1120), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1344)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 129024)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1568)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 1568), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 1792)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 1792), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2016)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 193536)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2240)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 2240), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2464)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 2464), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2688)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 258048)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 2912)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 2912), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3136)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 3136), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3360)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 322560)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3584)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 3584), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 3808)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 3808), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4032)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 387072)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4256)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 4256), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4480)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 4480), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4704)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 451584)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 4928)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 4928), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5152)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 5152), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5376)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 516096)]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5600)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 5600), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 32), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 2), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ kernel.shared_1[(threadIdx.x_2 + 5824)] = kernel_3[((((((floordiv(blockIdx.x, 7)*589824) + (floordiv((threadIdx.x_2 + 5824), 48)*4608)) + cse_var_2) + (floordiv(floormod((threadIdx.x_2 + 16), 48), 3)*9)) + cse_var_1) + floormod((threadIdx.x_2 + 1), 3))]
+ attr [IterVar(threadIdx.x_2, (nullptr), "ThreadIndex", "threadIdx.x")] "thread_extent" = 224;
+ if @tir.likely((threadIdx.x_2 < 96), dtype=bool) {
+ kernel.shared_1[(threadIdx.x_2 + 6048)] = kernel_3[(((((((floordiv(blockIdx.x, 7)*589824) + (floordiv(threadIdx.x_2, 48)*4608)) + cse_var_2) + (floordiv(floormod(threadIdx.x_2, 48), 3)*9)) + cse_var_1) + floormod(threadIdx.x_2, 3)) + 580608)]
+ }
+ for (rc.outer.inner: int32, 0, 2) {
+ for (rx.outer.inner: int32, 0, 3) {
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[(((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1536)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3072)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[(((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7))]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4608)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1539)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3075)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 9)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4611)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 6)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1542)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3078)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 18)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4614)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 9)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1545)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3081)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 27)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4617)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 12)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1548)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3084)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 36)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4620)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 15)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1551)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3087)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 45)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4623)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 18)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1554)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3090)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 54)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4626)]))
+ conv2d_nchw_1[0] = (conv2d_nchw_1[0] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 21)]))
+ conv2d_nchw_1[1] = (conv2d_nchw_1[1] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 1557)]))
+ conv2d_nchw_1[2] = (conv2d_nchw_1[2] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 3093)]))
+ conv2d_nchw_1[3] = (conv2d_nchw_1[3] + (pad_temp.shared_1[((((rc.outer.inner*72) + rx.outer.inner) + floormod(threadIdx.x, 7)) + 63)]*kernel.shared_1[((((floordiv(threadIdx.x, 7)*48) + (rc.outer.inner*24)) + rx.outer.inner) + 4629)]))
+ }
+ }
}
}
}
+ compute_3: Buffer(compute_2, float32, [25088], [])[((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7))] = max((conv2d_nchw_1[0] + bias_3: Buffer(bias_2, float32, [512], [])[((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7))]), 0f32)
+ compute_3[(((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7)) + 1568)] = max((conv2d_nchw_1[1] + bias_3[(((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7)) + 32)]), 0f32)
+ compute_3[(((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7)) + 3136)] = max((conv2d_nchw_1[2] + bias_3[(((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7)) + 64)]), 0f32)
+ compute_3[(((((floordiv(blockIdx.x, 7)*6272) + (floordiv(threadIdx.x, 7)*49)) + (floormod(blockIdx.x, 7)*7)) + floormod(threadIdx.x, 7)) + 4704)] = max((conv2d_nchw_1[3] + bias_3[(((floordiv(blockIdx.x, 7)*128) + floordiv(threadIdx.x, 7)) + 96)]), 0f32)
}
}
</pre></div>
@@ -1814,7 +657,7 @@ cooperative fetching, unrolling and operator fusion.</p>
<span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Execution time of this operator: 0.268 ms
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Execution time of this operator: 0.387 ms
</pre></div>
</div>
</div>
@@ -1844,35 +687,35 @@ conv2d_nchw_nn_o_o_i, conv2d_nchw_nn_o_i = s[conv2d_nchw].split(conv2d_nchw_nn_o
conv2d_nchw_nn_o_o_o_i, conv2d_nchw_nn_o_o_i = s[conv2d_nchw].split(conv2d_nchw_nn_o_o_i, factor=1)
conv2d_nchw_nn_o_o_o_o, conv2d_nchw_nn_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_nn_o_o_o_i, factor=1)
conv2d_nchw_ff_o_i, conv2d_nchw_ff_i = s[conv2d_nchw].split(conv2d_nchw_ff, factor=1)
-conv2d_nchw_ff_o_o_i, conv2d_nchw_ff_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_i, factor=2)
-conv2d_nchw_ff_o_o_o_i, conv2d_nchw_ff_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_i, factor=8)
-conv2d_nchw_ff_o_o_o_o, conv2d_nchw_ff_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_o_i, factor=2)
+conv2d_nchw_ff_o_o_i, conv2d_nchw_ff_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_i, factor=1)
+conv2d_nchw_ff_o_o_o_i, conv2d_nchw_ff_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_i, factor=32)
+conv2d_nchw_ff_o_o_o_o, conv2d_nchw_ff_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_ff_o_o_o_i, factor=4)
conv2d_nchw_yy_o_i, conv2d_nchw_yy_i = s[conv2d_nchw].split(conv2d_nchw_yy, factor=1)
conv2d_nchw_yy_o_o_i, conv2d_nchw_yy_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_i, factor=1)
-conv2d_nchw_yy_o_o_o_i, conv2d_nchw_yy_o_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_o_i, factor=7)
+conv2d_nchw_yy_o_o_o_i, conv2d_nchw_yy_o_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_o_i, factor=1)
conv2d_nchw_yy_o_o_o_o, conv2d_nchw_yy_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_yy_o_o_o_i, factor=1)
-conv2d_nchw_xx_o_i, conv2d_nchw_xx_i = s[conv2d_nchw].split(conv2d_nchw_xx, factor=7)
+conv2d_nchw_xx_o_i, conv2d_nchw_xx_i = s[conv2d_nchw].split(conv2d_nchw_xx, factor=1)
conv2d_nchw_xx_o_o_i, conv2d_nchw_xx_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_i, factor=1)
-conv2d_nchw_xx_o_o_o_i, conv2d_nchw_xx_o_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_o_i, factor=1)
+conv2d_nchw_xx_o_o_o_i, conv2d_nchw_xx_o_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_o_i, factor=7)
conv2d_nchw_xx_o_o_o_o, conv2d_nchw_xx_o_o_o_i = s[conv2d_nchw].split(conv2d_nchw_xx_o_o_o_i, factor=1)
-conv2d_nchw_rc_o_i, conv2d_nchw_rc_i = s[conv2d_nchw].split(conv2d_nchw_rc, factor=4)
-conv2d_nchw_rc_o_o, conv2d_nchw_rc_o_i = s[conv2d_nchw].split(conv2d_nchw_rc_o_i, factor=4)
-conv2d_nchw_ry_o_i, conv2d_nchw_ry_i = s[conv2d_nchw].split(conv2d_nchw_ry, factor=3)
+conv2d_nchw_rc_o_i, conv2d_nchw_rc_i = s[conv2d_nchw].split(conv2d_nchw_rc, factor=8)
+conv2d_nchw_rc_o_o, conv2d_nchw_rc_o_i = s[conv2d_nchw].split(conv2d_nchw_rc_o_i, factor=2)
+conv2d_nchw_ry_o_i, conv2d_nchw_ry_i = s[conv2d_nchw].split(conv2d_nchw_ry, factor=1)
conv2d_nchw_ry_o_o, conv2d_nchw_ry_o_i = s[conv2d_nchw].split(conv2d_nchw_ry_o_i, factor=1)
-conv2d_nchw_rx_o_i, conv2d_nchw_rx_i = s[conv2d_nchw].split(conv2d_nchw_rx, factor=3)
-conv2d_nchw_rx_o_o, conv2d_nchw_rx_o_i = s[conv2d_nchw].split(conv2d_nchw_rx_o_i, factor=1)
+conv2d_nchw_rx_o_i, conv2d_nchw_rx_i = s[conv2d_nchw].split(conv2d_nchw_rx, factor=1)
+conv2d_nchw_rx_o_o, conv2d_nchw_rx_o_i = s[conv2d_nchw].split(conv2d_nchw_rx_o_i, factor=3)
s[conv2d_nchw].reorder(conv2d_nchw_nn_o_o_o_o, conv2d_nchw_ff_o_o_o_o, conv2d_nchw_yy_o_o_o_o, conv2d_nchw_xx_o_o_o_o, conv2d_nchw_nn_o_o_o_i, conv2d_nchw_ff_o_o_o_i, conv2d_nchw_yy_o_o_o_i, conv2d_nchw_xx_o_o_o_i, conv2d_nchw_nn_o_o_i, conv2d_nchw_ff_o_o_i, conv2d_nchw_yy_o_o_i, conv2d_nchw_xx_o_o_i, conv2d_nchw_rc_o_o, conv2d_nchw_ry_o_o, conv2d_nchw_rx_o_o, conv2d_nchw_rc_o_i, conv2d_nchw_ry_o_i, conv2d_nchw_rx_o_i, conv2d_nchw_nn_o_i, conv2d_nchw_ff_o_i, conv2d_nchw_yy_o_i, conv2d_nc [...]
compute_i0_o_i, compute_i0_i = s[compute].split(compute_i0, factor=1)
compute_i0_o_o_i, compute_i0_o_i = s[compute].split(compute_i0_o_i, factor=1)
compute_i0_o_o_o, compute_i0_o_o_i = s[compute].split(compute_i0_o_o_i, factor=1)
-compute_i1_o_i, compute_i1_i = s[compute].split(compute_i1, factor=2)
-compute_i1_o_o_i, compute_i1_o_i = s[compute].split(compute_i1_o_i, factor=8)
-compute_i1_o_o_o, compute_i1_o_o_i = s[compute].split(compute_i1_o_o_i, factor=2)
+compute_i1_o_i, compute_i1_i = s[compute].split(compute_i1, factor=1)
+compute_i1_o_o_i, compute_i1_o_i = s[compute].split(compute_i1_o_i, factor=32)
+compute_i1_o_o_o, compute_i1_o_o_i = s[compute].split(compute_i1_o_o_i, factor=4)
compute_i2_o_i, compute_i2_i = s[compute].split(compute_i2, factor=1)
-compute_i2_o_o_i, compute_i2_o_i = s[compute].split(compute_i2_o_i, factor=7)
+compute_i2_o_o_i, compute_i2_o_i = s[compute].split(compute_i2_o_i, factor=1)
compute_i2_o_o_o, compute_i2_o_o_i = s[compute].split(compute_i2_o_o_i, factor=1)
-compute_i3_o_i, compute_i3_i = s[compute].split(compute_i3, factor=7)
-compute_i3_o_o_i, compute_i3_o_i = s[compute].split(compute_i3_o_i, factor=1)
+compute_i3_o_i, compute_i3_i = s[compute].split(compute_i3, factor=1)
+compute_i3_o_o_i, compute_i3_o_i = s[compute].split(compute_i3_o_i, factor=7)
compute_i3_o_o_o, compute_i3_o_o_i = s[compute].split(compute_i3_o_o_i, factor=1)
s[compute].reorder(compute_i0_o_o_o, compute_i1_o_o_o, compute_i2_o_o_o, compute_i3_o_o_o, compute_i0_o_o_i, compute_i1_o_o_i, compute_i2_o_o_i, compute_i3_o_o_i, compute_i0_o_i, compute_i1_o_i, compute_i2_o_i, compute_i3_o_i, compute_i0_i, compute_i1_i, compute_i2_i, compute_i3_i)
s[conv2d_nchw].compute_at(s[compute], compute_i3_o_i)
@@ -1892,14 +735,14 @@ s[compute].bind(compute_i0_o_i_i1_o_i_fused_i2_o_i_fused_i3_o_i_fused, te.thread
kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused = s[kernel_shared].fuse(kernel_shared_ax0, kernel_shared_ax1, kernel_shared_ax2, kernel_shared_ax3)
kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i = s[kernel_shared].split(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused, factor=1)
s[kernel_shared].vectorize(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i)
-kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[kernel_shared].split(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=56)
+kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[kernel_shared].split(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=224)
s[kernel_shared].bind(kernel_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i, te.thread_axis("threadIdx.x"))
pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused = s[pad_temp_shared].fuse(pad_temp_shared_ax0, pad_temp_shared_ax1, pad_temp_shared_ax2, pad_temp_shared_ax3)
pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i = s[pad_temp_shared].split(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused, factor=1)
s[pad_temp_shared].vectorize(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_i)
-pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[pad_temp_shared].split(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=56)
+pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_o, pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i = s[pad_temp_shared].split(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o, factor=224)
s[pad_temp_shared].bind(pad_temp_shared_ax0_ax1_fused_ax2_fused_ax3_fused_o_i, te.thread_axis("threadIdx.x"))
-s[conv2d_nchw].pragma(conv2d_nchw_nn_o_o_o_o, "auto_unroll_max_step", 1024)
+s[conv2d_nchw].pragma(conv2d_nchw_nn_o_o_o_o, "auto_unroll_max_step", 64)
s[conv2d_nchw].pragma(conv2d_nchw_nn_o_o_o_o, "unroll_explicit", True)
CUDA source code:
@@ -1917,1169 +760,93 @@ CUDA source code:
#define int64_t long long
#define uint64_t unsigned long long
#endif
-extern "C" __global__ void __launch_bounds__(56) default_function_kernel0(float* __restrict__ data, float* __restrict__ kernel, float* __restrict__ compute, float* __restrict__ bias) {
- float conv2d_nchw[28];
- __shared__ float pad_temp_shared[1296];
- __shared__ float kernel_shared[4608];
+extern "C" __global__ void __launch_bounds__(224) default_function_kernel0(float* __restrict__ data, float* __restrict__ kernel, float* __restrict__ compute, float* __restrict__ bias) {
+ float conv2d_nchw[4];
+ __shared__ float pad_temp_shared[144];
+ __shared__ float kernel_shared[6144];
conv2d_nchw[0] = 0.000000e+00f;
- conv2d_nchw[14] = 0.000000e+00f;
conv2d_nchw[1] = 0.000000e+00f;
- conv2d_nchw[15] = 0.000000e+00f;
conv2d_nchw[2] = 0.000000e+00f;
- conv2d_nchw[16] = 0.000000e+00f;
conv2d_nchw[3] = 0.000000e+00f;
- conv2d_nchw[17] = 0.000000e+00f;
- conv2d_nchw[4] = 0.000000e+00f;
- conv2d_nchw[18] = 0.000000e+00f;
- conv2d_nchw[5] = 0.000000e+00f;
- conv2d_nchw[19] = 0.000000e+00f;
- conv2d_nchw[6] = 0.000000e+00f;
- conv2d_nchw[20] = 0.000000e+00f;
- conv2d_nchw[7] = 0.000000e+00f;
- conv2d_nchw[21] = 0.000000e+00f;
- conv2d_nchw[8] = 0.000000e+00f;
- conv2d_nchw[22] = 0.000000e+00f;
- conv2d_nchw[9] = 0.000000e+00f;
- conv2d_nchw[23] = 0.000000e+00f;
- conv2d_nchw[10] = 0.000000e+00f;
- conv2d_nchw[24] = 0.000000e+00f;
- conv2d_nchw[11] = 0.000000e+00f;
- conv2d_nchw[25] = 0.000000e+00f;
- conv2d_nchw[12] = 0.000000e+00f;
- conv2d_nchw[26] = 0.000000e+00f;
- conv2d_nchw[13] = 0.000000e+00f;
- conv2d_nchw[27] = 0.000000e+00f;
for (int rc_outer_outer = 0; rc_outer_outer < 32; ++rc_outer_outer) {
- __syncthreads();
- pad_temp_shared[((int)threadIdx.x)] = ((((9 <= ((int)threadIdx.x)) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[((((rc_outer_outer * 784) + ((((int)threadIdx.x) / 9) * 7)) + (((int)threadIdx.x) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 56)] = (((((9 <= ((((int)threadIdx.x) + 56) % 81)) && (((((int)threadIdx.x) + 56) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 2) % 9))) && (((((int)threadIdx.x) + 2) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 56) / 81) * 49)) + ((((((int)threadIdx.x) + 56) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 2) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 112)] = (((((9 <= ((((int)threadIdx.x) + 31) % 81)) && (((((int)threadIdx.x) + 31) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 4) % 9))) && (((((int)threadIdx.x) + 4) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 112) / 81) * 49)) + ((((((int)threadIdx.x) + 31) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 4) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 168)] = ((((3 <= ((int)threadIdx.x)) && (1 <= ((((int)threadIdx.x) + 6) % 9))) && (((((int)threadIdx.x) + 6) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 168) / 81) * 49)) + (((((int)threadIdx.x) + 6) / 9) * 7)) + ((((int)threadIdx.x) + 6) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 224)] = (((((9 <= ((((int)threadIdx.x) + 62) % 81)) && (((((int)threadIdx.x) + 62) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 8) % 9))) && (((((int)threadIdx.x) + 8) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 224) / 81) * 49)) + ((((((int)threadIdx.x) + 62) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 8) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 280)] = (((((9 <= ((((int)threadIdx.x) + 37) % 81)) && (((((int)threadIdx.x) + 37) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 1) % 9))) && (((((int)threadIdx.x) + 1) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 280) / 81) * 49)) + ((((((int)threadIdx.x) + 37) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 1) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 336)] = (((1 <= ((((int)threadIdx.x) + 3) % 9)) && (((((int)threadIdx.x) + 3) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 336) / 81) * 49)) + (((((int)threadIdx.x) + 12) / 9) * 7)) + ((((int)threadIdx.x) + 3) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 392)] = (((((9 <= ((((int)threadIdx.x) + 68) % 81)) && (((((int)threadIdx.x) + 68) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 5) % 9))) && (((((int)threadIdx.x) + 5) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 392) / 81) * 49)) + ((((((int)threadIdx.x) + 68) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 5) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 448)] = (((((9 <= ((((int)threadIdx.x) + 43) % 81)) && (((((int)threadIdx.x) + 43) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 7) % 9))) && (((((int)threadIdx.x) + 7) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 448) / 81) * 49)) + ((((((int)threadIdx.x) + 43) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 7) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 504)] = ((((((int)threadIdx.x) < 54) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 504) / 81) * 49)) + ((((int)threadIdx.x) / 9) * 7)) + (((int)threadIdx.x) % 9)) + 6)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 560)] = (((((9 <= ((((int)threadIdx.x) + 74) % 81)) && (((((int)threadIdx.x) + 74) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 2) % 9))) && (((((int)threadIdx.x) + 2) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 560) / 81) * 49)) + ((((((int)threadIdx.x) + 74) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 2) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 616)] = (((((9 <= ((((int)threadIdx.x) + 49) % 81)) && (((((int)threadIdx.x) + 49) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 4) % 9))) && (((((int)threadIdx.x) + 4) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 616) / 81) * 49)) + ((((((int)threadIdx.x) + 49) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 4) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 672)] = ((((((int)threadIdx.x) < 48) && (1 <= ((((int)threadIdx.x) + 6) % 9))) && (((((int)threadIdx.x) + 6) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 672) / 81) * 49)) + (((((int)threadIdx.x) + 24) / 9) * 7)) + ((((int)threadIdx.x) + 6) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 728)] = (((((9 <= ((((int)threadIdx.x) + 80) % 81)) && (((((int)threadIdx.x) + 80) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 8) % 9))) && (((((int)threadIdx.x) + 8) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 728) / 81) * 49)) + ((((((int)threadIdx.x) + 80) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 8) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 784)] = (((((9 <= ((((int)threadIdx.x) + 55) % 81)) && (((((int)threadIdx.x) + 55) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 1) % 9))) && (((((int)threadIdx.x) + 1) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 784) / 81) * 49)) + ((((((int)threadIdx.x) + 55) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 1) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 840)] = (((((9 <= ((((int)threadIdx.x) + 30) % 81)) && (((((int)threadIdx.x) + 30) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 3) % 9))) && (((((int)threadIdx.x) + 3) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 840) / 81) * 49)) + ((((((int)threadIdx.x) + 30) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 3) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 896)] = ((((4 <= ((int)threadIdx.x)) && (1 <= ((((int)threadIdx.x) + 5) % 9))) && (((((int)threadIdx.x) + 5) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 896) / 81) * 49)) + (((((int)threadIdx.x) + 5) / 9) * 7)) + ((((int)threadIdx.x) + 5) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 952)] = (((((9 <= ((((int)threadIdx.x) + 61) % 81)) && (((((int)threadIdx.x) + 61) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 7) % 9))) && (((((int)threadIdx.x) + 7) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 952) / 81) * 49)) + ((((((int)threadIdx.x) + 61) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 7) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1008)] = (((((1 <= (((((int)threadIdx.x) / 9) + 4) % 9)) && (((((int)threadIdx.x) + 36) % 81) < 72)) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1008) / 81) * 49)) + ((((((int)threadIdx.x) / 9) + 4) % 9) * 7)) + (((int)threadIdx.x) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1064)] = (((1 <= ((((int)threadIdx.x) + 2) % 9)) && (((((int)threadIdx.x) + 2) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1064) / 81) * 49)) + (((((int)threadIdx.x) + 11) / 9) * 7)) + ((((int)threadIdx.x) + 2) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1120)] = (((((9 <= ((((int)threadIdx.x) + 67) % 81)) && (((((int)threadIdx.x) + 67) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 4) % 9))) && (((((int)threadIdx.x) + 4) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1120) / 81) * 49)) + ((((((int)threadIdx.x) + 67) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 4) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1176)] = (((((9 <= ((((int)threadIdx.x) + 42) % 81)) && (((((int)threadIdx.x) + 42) % 81) < 72)) && (1 <= ((((int)threadIdx.x) + 6) % 9))) && (((((int)threadIdx.x) + 6) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1176) / 81) * 49)) + ((((((int)threadIdx.x) + 42) % 81) / 9) * 7)) + ((((int)threadIdx.x) + 6) % 9)) - 8)] : 0.000000e+00f);
- pad_temp_shared[(((int)threadIdx.x) + 1232)] = ((((((int)threadIdx.x) < 55) && (1 <= ((((int)threadIdx.x) + 8) % 9))) && (((((int)threadIdx.x) + 8) % 9) < 8)) ? data[(((((rc_outer_outer * 784) + (((((int)threadIdx.x) + 1232) / 81) * 49)) + (((((int)threadIdx.x) + 17) / 9) * 7)) + ((((int)threadIdx.x) + 8) % 9)) - 8)] : 0.000000e+00f);
- if (((int)threadIdx.x) < 8) {
- pad_temp_shared[(((int)threadIdx.x) + 1288)] = 0.000000e+00f;
- }
- kernel_shared[((int)threadIdx.x)] = kernel[(((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x))];
- kernel_shared[(((int)threadIdx.x) + 56)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 112)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 112) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 168)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 168) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 224)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 224) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 280)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 280) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 336)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 336) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 392)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 392) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 448)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 448) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 504)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 504) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 560)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 560) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 616)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 616) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 672)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 672) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 728)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 728) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 784)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 784) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 840)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 840) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 896)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 896) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 952)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 952) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1008)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 32256)];
- kernel_shared[(((int)threadIdx.x) + 1064)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1064) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1120)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1120) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1176)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1176) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 1232)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1232) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1288)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1288) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1344)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1344) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 1400)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1400) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1456)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1456) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1512)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1512) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 1568)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1568) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1624)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1624) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1680)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1680) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1736)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1736) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1792)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1792) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1848)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1848) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1904)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1904) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 1960)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 1960) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2016)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 64512)];
- kernel_shared[(((int)threadIdx.x) + 2072)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2072) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2128)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2128) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2184)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2184) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 2240)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2240) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2296)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2296) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2352)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2352) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 2408)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2408) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2464)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2464) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2520)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2520) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 2576)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2576) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2632)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2632) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2688)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2688) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2744)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2744) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2800)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2800) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2856)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2856) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2912)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2912) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 2968)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 2968) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3024)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 96768)];
- kernel_shared[(((int)threadIdx.x) + 3080)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3080) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3136)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3136) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3192)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3192) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 3248)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3248) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3304)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3304) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3360)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3360) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 3416)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3416) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3472)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3472) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3528)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3528) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- kernel_shared[(((int)threadIdx.x) + 3584)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3584) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 128) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3640)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3640) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 40) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3696)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3696) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 32) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3752)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3752) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 8) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3808)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3808) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 64) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3864)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3864) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) / 3) + 40) % 48) * 3)) + (((int)threadIdx.x) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3920)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3920) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 32) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 3976)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 3976) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 88) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4032)] = kernel[((((((int)blockIdx.x) * 147456) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 129024)];
- kernel_shared[(((int)threadIdx.x) + 4088)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4088) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 56) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4144)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4144) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 112) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4200)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4200) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 24)];
- kernel_shared[(((int)threadIdx.x) + 4256)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4256) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 80) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4312)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4312) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 136) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4368)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4368) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 48)];
- kernel_shared[(((int)threadIdx.x) + 4424)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4424) / 144) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 104) % 144) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4480)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4480) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 16) / 3) * 3)) + ((((int)threadIdx.x) + 1) % 3))];
- kernel_shared[(((int)threadIdx.x) + 4536)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4536) / 144) * 4608)) + (rc_outer_outer * 144)) + ((int)threadIdx.x)) + 72)];
- if (((int)threadIdx.x) < 16) {
- kernel_shared[(((int)threadIdx.x) + 4592)] = kernel[(((((((int)blockIdx.x) * 147456) + (((((int)threadIdx.x) + 4592) / 144) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) + 128) / 3) * 3)) + ((((int)threadIdx.x) + 2) % 3))];
- }
- __syncthreads();
- for (int rc_outer_inner = 0; rc_outer_inner < 4; ++rc_outer_inner) {
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[(((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36))]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2304)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 1)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2305)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2306)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 3)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2307)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 4)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2308)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 5)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2309)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 6)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2310)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 7)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2311)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 8)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2312)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 9)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2313)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 10)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2314)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 11)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2315)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 12)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2316)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 13)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2317)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 14)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2318)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 15)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2319)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 16)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2320)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 17)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2321)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 18)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2322)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 19)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2323)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 20)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2324)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 21)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2325)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 22)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2326)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 23)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2327)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 24)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2328)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 25)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2329)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 26)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2330)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 27)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2331)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 28)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2332)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 29)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2333)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 30)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2334)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 31)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2335)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 32)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2336)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 33)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2337)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 34)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2338)]));
- conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[14] = (conv2d_nchw[14] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[15] = (conv2d_nchw[15] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[16] = (conv2d_nchw[16] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[17] = (conv2d_nchw[17] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[4] = (conv2d_nchw[4] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[18] = (conv2d_nchw[18] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[5] = (conv2d_nchw[5] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[19] = (conv2d_nchw[19] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[6] = (conv2d_nchw[6] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 35)]));
- conv2d_nchw[20] = (conv2d_nchw[20] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2339)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9))] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 144)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2448)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 1)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 145)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2449)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 2)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 3)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 4)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 5)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 6)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 7)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 146)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 8)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2450)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 9)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 147)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2451)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 10)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 148)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2452)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 11)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 12)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 13)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 14)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 15)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 16)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 149)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 17)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2453)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 18)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 150)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2454)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 19)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 151)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2455)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 20)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 21)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 22)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 23)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 24)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 25)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 152)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 26)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2456)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 81)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 153)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2457)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 82)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 154)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2458)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 83)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 84)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 85)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 86)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 87)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 88)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 155)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 89)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2459)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 90)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 156)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2460)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 91)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 157)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2461)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 92)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 93)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 94)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 95)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 96)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 97)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 158)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 98)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2462)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 99)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 159)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2463)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 100)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 160)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2464)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 101)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 102)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 103)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 104)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 105)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 106)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 161)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 107)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2465)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 162)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 162)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2466)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 163)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 163)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2467)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 164)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 165)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 166)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 167)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 168)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 169)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 164)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 170)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2468)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 171)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 165)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2469)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 172)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 166)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2470)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 173)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 174)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 175)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 176)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 177)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 178)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 167)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 179)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2471)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 180)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 168)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2472)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 181)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 169)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2473)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 182)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 183)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 184)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 185)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 186)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 187)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 170)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 188)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2474)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 243)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 171)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2475)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 244)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 172)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2476)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 245)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 246)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 247)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 248)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 249)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 250)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 173)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 251)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2477)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 252)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 174)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2478)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 253)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 175)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2479)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 254)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 255)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 256)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 257)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 258)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 259)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 176)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 260)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2480)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 261)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 177)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2481)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 262)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 178)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2482)]));
- conv2d_nchw[7] = (conv2d_nchw[7] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[21] = (conv2d_nchw[21] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 263)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[8] = (conv2d_nchw[8] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[22] = (conv2d_nchw[22] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 264)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[9] = (conv2d_nchw[9] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[23] = (conv2d_nchw[23] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 265)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[10] = (conv2d_nchw[10] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[24] = (conv2d_nchw[24] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 266)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[11] = (conv2d_nchw[11] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[25] = (conv2d_nchw[25] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 267)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[12] = (conv2d_nchw[12] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[26] = (conv2d_nchw[26] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 268)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- conv2d_nchw[13] = (conv2d_nchw[13] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 179)]));
- conv2d_nchw[27] = (conv2d_nchw[27] + (pad_temp_shared[(((rc_outer_inner * 324) + ((((int)threadIdx.x) % 7) * 9)) + 269)] * kernel_shared[((((((int)threadIdx.x) / 7) * 288) + (rc_outer_inner * 36)) + 2483)]));
- }
- }
- for (int i1_inner = 0; i1_inner < 2; ++i1_inner) {
- for (int i3_inner = 0; i3_inner < 7; ++i3_inner) {
- compute[(((((((int)blockIdx.x) * 1568) + ((((int)threadIdx.x) / 7) * 98)) + (i1_inner * 49)) + ((((int)threadIdx.x) % 7) * 7)) + i3_inner)] = max((conv2d_nchw[((i1_inner * 7) + i3_inner)] + bias[(((((int)blockIdx.x) * 32) + ((((int)threadIdx.x) / 7) * 2)) + i1_inner)]), 0.000000e+00f);
- compute[((((((((int)blockIdx.x) * 1568) + ((((int)threadIdx.x) / 7) * 98)) + (i1_inner * 49)) + ((((int)threadIdx.x) % 7) * 7)) + i3_inner) + 784)] = max((conv2d_nchw[(((i1_inner * 7) + i3_inner) + 14)] + bias[((((((int)blockIdx.x) * 32) + ((((int)threadIdx.x) / 7) * 2)) + i1_inner) + 16)]), 0.000000e+00f);
+ for (int ry_outer_outer = 0; ry_outer_outer < 3; ++ry_outer_outer) {
+ __syncthreads();
+ if (((int)threadIdx.x) < 144) {
+ pad_temp_shared[((int)threadIdx.x)] = (((((1 <= (ry_outer_outer + (((int)blockIdx.x) % 7))) && ((ry_outer_outer + (((int)blockIdx.x) % 7)) < 8)) && (1 <= (((int)threadIdx.x) % 9))) && ((((int)threadIdx.x) % 9) < 8)) ? data[((((((rc_outer_outer * 784) + ((((int)threadIdx.x) / 9) * 49)) + (ry_outer_outer * 7)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 9)) - 8)] : 0.000000e+00f);
+ }
+ kernel_shared[((int)threadIdx.x)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 224)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 224) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 448)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 448) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 672)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 64512)];
+ kernel_shared[(((int)threadIdx.x) + 896)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 896) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 1120)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 1120) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 1344)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 129024)];
+ kernel_shared[(((int)threadIdx.x) + 1568)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 1568) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 1792)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 1792) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 2016)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 193536)];
+ kernel_shared[(((int)threadIdx.x) + 2240)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 2240) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 2464)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 2464) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 2688)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 258048)];
+ kernel_shared[(((int)threadIdx.x) + 2912)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 2912) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 3136)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 3136) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 3360)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 322560)];
+ kernel_shared[(((int)threadIdx.x) + 3584)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 3584) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 3808)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 3808) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 4032)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 387072)];
+ kernel_shared[(((int)threadIdx.x) + 4256)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 4256) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 4480)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 4480) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 4704)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 451584)];
+ kernel_shared[(((int)threadIdx.x) + 4928)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 4928) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 5152)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 5152) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 5376)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 516096)];
+ kernel_shared[(((int)threadIdx.x) + 5600)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 5600) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 32) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 2) % 3))];
+ kernel_shared[(((int)threadIdx.x) + 5824)] = kernel[(((((((((int)blockIdx.x) / 7) * 589824) + (((((int)threadIdx.x) + 5824) / 48) * 4608)) + (rc_outer_outer * 144)) + ((((((int)threadIdx.x) + 16) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + ((((int)threadIdx.x) + 1) % 3))];
+ if (((int)threadIdx.x) < 96) {
+ kernel_shared[(((int)threadIdx.x) + 6048)] = kernel[((((((((((int)blockIdx.x) / 7) * 589824) + ((((int)threadIdx.x) / 48) * 4608)) + (rc_outer_outer * 144)) + (((((int)threadIdx.x) % 48) / 3) * 9)) + (ry_outer_outer * 3)) + (((int)threadIdx.x) % 3)) + 580608)];
+ }
+ __syncthreads();
+ for (int rc_outer_inner = 0; rc_outer_inner < 2; ++rc_outer_inner) {
+ for (int rx_outer_inner = 0; rx_outer_inner < 3; ++rx_outer_inner) {
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1536)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3072)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[(((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7))] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4608)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1539)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3075)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 9)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4611)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 6)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1542)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3078)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 18)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4614)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 9)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1545)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3081)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 27)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4617)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 12)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1548)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3084)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 36)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4620)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 15)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1551)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3087)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 45)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4623)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 18)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1554)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3090)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 54)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4626)]));
+ conv2d_nchw[0] = (conv2d_nchw[0] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 21)]));
+ conv2d_nchw[1] = (conv2d_nchw[1] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 1557)]));
+ conv2d_nchw[2] = (conv2d_nchw[2] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 3093)]));
+ conv2d_nchw[3] = (conv2d_nchw[3] + (pad_temp_shared[((((rc_outer_inner * 72) + rx_outer_inner) + (((int)threadIdx.x) % 7)) + 63)] * kernel_shared[(((((((int)threadIdx.x) / 7) * 48) + (rc_outer_inner * 24)) + rx_outer_inner) + 4629)]));
+ }
+ }
}
}
+ compute[(((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7))] = max((conv2d_nchw[0] + bias[(((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7))]), 0.000000e+00f);
+ compute[((((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7)) + 1568)] = max((conv2d_nchw[1] + bias[((((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7)) + 32)]), 0.000000e+00f);
+ compute[((((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7)) + 3136)] = max((conv2d_nchw[2] + bias[((((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7)) + 64)]), 0.000000e+00f);
+ compute[((((((((int)blockIdx.x) / 7) * 6272) + ((((int)threadIdx.x) / 7) * 49)) + ((((int)blockIdx.x) % 7) * 7)) + (((int)threadIdx.x) % 7)) + 4704)] = max((conv2d_nchw[3] + bias[((((((int)blockIdx.x) / 7) * 128) + (((int)threadIdx.x) / 7)) + 96)]), 0.000000e+00f);
}
</pre></div>
</div>
@@ -3115,7 +882,7 @@ In the example below we resume the status and do more 5 trials.</p>
Get devices for measurement successfully!
</pre></div>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 5 minutes 42.541 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 5 minutes 27.158 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-tune-with-autoscheduler-tune-conv2d-layer-cuda-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/e3e540f3b477c0c52d8eb73e674e8ffd/tune_conv2d_layer_cuda.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">tune_conv2d_layer_cuda.py</span></code></a></p>
diff --git a/docs/how_to/tune_with_autoscheduler/tune_network_cuda.html b/docs/how_to/tune_with_autoscheduler/tune_network_cuda.html
index 15519eefd1..1977ea2a46 100644
--- a/docs/how_to/tune_with_autoscheduler/tune_network_cuda.html
+++ b/docs/how_to/tune_with_autoscheduler/tune_network_cuda.html
@@ -915,7 +915,7 @@ so we can read the log file and load the best schedules.</p>
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 7.8418 7.8454 7.8478 7.8321 0.0069
+ 7.9013 7.9087 7.9108 7.8845 0.0119
</pre></div>
</div>
</div>
@@ -937,7 +937,7 @@ to learn how to use the RPC Tracker and RPC Server.
To use the RPC Tracker in auto-scheduler, replace the runner in <code class="code docutils literal notranslate"><span class="pre">TuningOptions</span></code>
with <a class="reference internal" href="../../reference/api/python/auto_scheduler.html#tvm.auto_scheduler.RPCRunner" title="tvm.auto_scheduler.RPCRunner"><code class="xref any py py-class docutils literal notranslate"><span class="pre">auto_scheduler.RPCRunner</span></code></a>.</p></li>
</ol>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 1.296 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 1.762 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-tune-with-autoscheduler-tune-network-cuda-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/eafe360d52540634c9eea0fa89e804bd/tune_network_cuda.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">tune_network_cuda.py</span></code></a></p>
diff --git a/docs/how_to/tune_with_autoscheduler/tune_network_x86.html b/docs/how_to/tune_with_autoscheduler/tune_network_x86.html
index 539818af59..c246e534ae 100644
--- a/docs/how_to/tune_with_autoscheduler/tune_network_x86.html
+++ b/docs/how_to/tune_with_autoscheduler/tune_network_x86.html
@@ -934,7 +934,7 @@ so we can read the log file and load the best schedules.</p>
Evaluate inference time cost...
Execution time summary:
mean (ms) median (ms) max (ms) min (ms) std (ms)
- 766.8676 768.5019 769.7074 762.3936 3.2017
+ 751.4864 753.3622 753.5823 747.5147 2.8099
</pre></div>
</div>
</div>
@@ -956,7 +956,7 @@ to learn how to use the RPC Tracker and RPC Server.
To use the RPC Tracker in auto-scheduler, replace the runner in <code class="code docutils literal notranslate"><span class="pre">TuningOptions</span></code>
with <a class="reference internal" href="../../reference/api/python/auto_scheduler.html#tvm.auto_scheduler.RPCRunner" title="tvm.auto_scheduler.RPCRunner"><code class="xref any py py-class docutils literal notranslate"><span class="pre">auto_scheduler.RPCRunner</span></code></a>.</p></li>
</ol>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 31.335 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 31.258 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-tune-with-autoscheduler-tune-network-x86-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/e416b94ca1090b0897c0f6e0df95b911/tune_network_x86.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">tune_network_x86.py</span></code></a></p>
diff --git a/docs/how_to/tune_with_autoscheduler/tune_sparse_x86.html b/docs/how_to/tune_with_autoscheduler/tune_sparse_x86.html
index a2bfcccc25..afa4ac50ef 100644
--- a/docs/how_to/tune_with_autoscheduler/tune_sparse_x86.html
+++ b/docs/how_to/tune_with_autoscheduler/tune_sparse_x86.html
@@ -632,337 +632,27 @@ layout transformation, parallelization, vectorization, unrolling, and operator f
placeholder_4: Buffer(placeholder_14: Pointer(float32), float32, [128, 512], []),
compute: Buffer(compute_2: Pointer(float32), float32, [128, 512], [])}
buffer_map = {placeholder_5: placeholder, placeholder_6: placeholder_1, placeholder_7: placeholder_2, placeholder_8: placeholder_3, placeholder_9: placeholder_4, compute_1: compute} {
- for (i0.outer.i1.outer.fused: int32, 0, 32) "parallel" {
- allocate(compute_3: Pointer(global float32), float32, [2048]), storage_scope = global {
- for (i.outer.inner: int32, 0, 32) {
- let cse_var_1: int32 = (i.outer.inner*64)
- {
- compute_4: Buffer(compute_3, float32, [2048], [])[cse_var_1] = 0f32
- compute_4[(cse_var_1 + 1)] = 0f32
- compute_4[(cse_var_1 + 2)] = 0f32
- compute_4[(cse_var_1 + 3)] = 0f32
- compute_4[(cse_var_1 + 4)] = 0f32
- compute_4[(cse_var_1 + 5)] = 0f32
- compute_4[(cse_var_1 + 6)] = 0f32
- compute_4[(cse_var_1 + 7)] = 0f32
- compute_4[(cse_var_1 + 8)] = 0f32
- compute_4[(cse_var_1 + 9)] = 0f32
- compute_4[(cse_var_1 + 10)] = 0f32
- compute_4[(cse_var_1 + 11)] = 0f32
- compute_4[(cse_var_1 + 12)] = 0f32
- compute_4[(cse_var_1 + 13)] = 0f32
- compute_4[(cse_var_1 + 14)] = 0f32
- compute_4[(cse_var_1 + 15)] = 0f32
- compute_4[(cse_var_1 + 16)] = 0f32
- compute_4[(cse_var_1 + 17)] = 0f32
- compute_4[(cse_var_1 + 18)] = 0f32
- compute_4[(cse_var_1 + 19)] = 0f32
- compute_4[(cse_var_1 + 20)] = 0f32
- compute_4[(cse_var_1 + 21)] = 0f32
- compute_4[(cse_var_1 + 22)] = 0f32
- compute_4[(cse_var_1 + 23)] = 0f32
- compute_4[(cse_var_1 + 24)] = 0f32
- compute_4[(cse_var_1 + 25)] = 0f32
- compute_4[(cse_var_1 + 26)] = 0f32
- compute_4[(cse_var_1 + 27)] = 0f32
- compute_4[(cse_var_1 + 28)] = 0f32
- compute_4[(cse_var_1 + 29)] = 0f32
- compute_4[(cse_var_1 + 30)] = 0f32
- compute_4[(cse_var_1 + 31)] = 0f32
- compute_4[(cse_var_1 + 32)] = 0f32
- compute_4[(cse_var_1 + 33)] = 0f32
- compute_4[(cse_var_1 + 34)] = 0f32
- compute_4[(cse_var_1 + 35)] = 0f32
- compute_4[(cse_var_1 + 36)] = 0f32
- compute_4[(cse_var_1 + 37)] = 0f32
- compute_4[(cse_var_1 + 38)] = 0f32
- compute_4[(cse_var_1 + 39)] = 0f32
- compute_4[(cse_var_1 + 40)] = 0f32
- compute_4[(cse_var_1 + 41)] = 0f32
- compute_4[(cse_var_1 + 42)] = 0f32
- compute_4[(cse_var_1 + 43)] = 0f32
- compute_4[(cse_var_1 + 44)] = 0f32
- compute_4[(cse_var_1 + 45)] = 0f32
- compute_4[(cse_var_1 + 46)] = 0f32
- compute_4[(cse_var_1 + 47)] = 0f32
- compute_4[(cse_var_1 + 48)] = 0f32
- compute_4[(cse_var_1 + 49)] = 0f32
- compute_4[(cse_var_1 + 50)] = 0f32
- compute_4[(cse_var_1 + 51)] = 0f32
- compute_4[(cse_var_1 + 52)] = 0f32
- compute_4[(cse_var_1 + 53)] = 0f32
- compute_4[(cse_var_1 + 54)] = 0f32
- compute_4[(cse_var_1 + 55)] = 0f32
- compute_4[(cse_var_1 + 56)] = 0f32
- compute_4[(cse_var_1 + 57)] = 0f32
- compute_4[(cse_var_1 + 58)] = 0f32
- compute_4[(cse_var_1 + 59)] = 0f32
- compute_4[(cse_var_1 + 60)] = 0f32
- compute_4[(cse_var_1 + 61)] = 0f32
- compute_4[(cse_var_1 + 62)] = 0f32
- compute_4[(cse_var_1 + 63)] = 0f32
- for (elem_idx: int32, 0, (placeholder_15: Buffer(placeholder_13, int32, [33], [])[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])) {
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- compute_4[cse_var_1] = (compute_4[cse_var_1] + (placeholder_16: Buffer(placeholder_11, float32, [78656], [])[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17: Buffer(placeholder_10, float32, [32768], [])[((i.outer.inner*1024) + placeholder_18: Buffer(placeholder_12, int32, [4916], [])[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_2: int32 = (cse_var_1 + 1)
- compute_4[cse_var_2] = (compute_4[cse_var_2] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_3: int32 = (cse_var_1 + 2)
- compute_4[cse_var_3] = (compute_4[cse_var_3] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_4: int32 = (cse_var_1 + 3)
- compute_4[cse_var_4] = (compute_4[cse_var_4] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_5: int32 = (cse_var_1 + 4)
- compute_4[cse_var_5] = (compute_4[cse_var_5] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_6: int32 = (cse_var_1 + 5)
- compute_4[cse_var_6] = (compute_4[cse_var_6] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_7: int32 = (cse_var_1 + 6)
- compute_4[cse_var_7] = (compute_4[cse_var_7] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_8: int32 = (cse_var_1 + 7)
- compute_4[cse_var_8] = (compute_4[cse_var_8] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_9: int32 = (cse_var_1 + 8)
- compute_4[cse_var_9] = (compute_4[cse_var_9] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_10: int32 = (cse_var_1 + 9)
- compute_4[cse_var_10] = (compute_4[cse_var_10] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_11: int32 = (cse_var_1 + 10)
- compute_4[cse_var_11] = (compute_4[cse_var_11] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_12: int32 = (cse_var_1 + 11)
- compute_4[cse_var_12] = (compute_4[cse_var_12] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_13: int32 = (cse_var_1 + 12)
- compute_4[cse_var_13] = (compute_4[cse_var_13] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_14: int32 = (cse_var_1 + 13)
- compute_4[cse_var_14] = (compute_4[cse_var_14] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_15: int32 = (cse_var_1 + 14)
- compute_4[cse_var_15] = (compute_4[cse_var_15] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_16: int32 = (cse_var_1 + 15)
- compute_4[cse_var_16] = (compute_4[cse_var_16] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)])], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_17: int32 = (cse_var_1 + 16)
- compute_4[cse_var_17] = (compute_4[cse_var_17] + (placeholder_16[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_18: int32 = (cse_var_1 + 17)
- compute_4[cse_var_18] = (compute_4[cse_var_18] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_19: int32 = (cse_var_1 + 18)
- compute_4[cse_var_19] = (compute_4[cse_var_19] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_20: int32 = (cse_var_1 + 19)
- compute_4[cse_var_20] = (compute_4[cse_var_20] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_21: int32 = (cse_var_1 + 20)
- compute_4[cse_var_21] = (compute_4[cse_var_21] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_22: int32 = (cse_var_1 + 21)
- compute_4[cse_var_22] = (compute_4[cse_var_22] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_23: int32 = (cse_var_1 + 22)
- compute_4[cse_var_23] = (compute_4[cse_var_23] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_24: int32 = (cse_var_1 + 23)
- compute_4[cse_var_24] = (compute_4[cse_var_24] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_25: int32 = (cse_var_1 + 24)
- compute_4[cse_var_25] = (compute_4[cse_var_25] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_26: int32 = (cse_var_1 + 25)
- compute_4[cse_var_26] = (compute_4[cse_var_26] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_27: int32 = (cse_var_1 + 26)
- compute_4[cse_var_27] = (compute_4[cse_var_27] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_28: int32 = (cse_var_1 + 27)
- compute_4[cse_var_28] = (compute_4[cse_var_28] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_29: int32 = (cse_var_1 + 28)
- compute_4[cse_var_29] = (compute_4[cse_var_29] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_30: int32 = (cse_var_1 + 29)
- compute_4[cse_var_30] = (compute_4[cse_var_30] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_31: int32 = (cse_var_1 + 30)
- compute_4[cse_var_31] = (compute_4[cse_var_31] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_32: int32 = (cse_var_1 + 31)
- compute_4[cse_var_32] = (compute_4[cse_var_32] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 256)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_33: int32 = (cse_var_1 + 32)
- compute_4[cse_var_33] = (compute_4[cse_var_33] + (placeholder_16[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_34: int32 = (cse_var_1 + 33)
- compute_4[cse_var_34] = (compute_4[cse_var_34] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_35: int32 = (cse_var_1 + 34)
- compute_4[cse_var_35] = (compute_4[cse_var_35] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_36: int32 = (cse_var_1 + 35)
- compute_4[cse_var_36] = (compute_4[cse_var_36] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_37: int32 = (cse_var_1 + 36)
- compute_4[cse_var_37] = (compute_4[cse_var_37] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_38: int32 = (cse_var_1 + 37)
- compute_4[cse_var_38] = (compute_4[cse_var_38] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_39: int32 = (cse_var_1 + 38)
- compute_4[cse_var_39] = (compute_4[cse_var_39] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_40: int32 = (cse_var_1 + 39)
- compute_4[cse_var_40] = (compute_4[cse_var_40] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_41: int32 = (cse_var_1 + 40)
- compute_4[cse_var_41] = (compute_4[cse_var_41] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_42: int32 = (cse_var_1 + 41)
- compute_4[cse_var_42] = (compute_4[cse_var_42] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_43: int32 = (cse_var_1 + 42)
- compute_4[cse_var_43] = (compute_4[cse_var_43] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_44: int32 = (cse_var_1 + 43)
- compute_4[cse_var_44] = (compute_4[cse_var_44] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_45: int32 = (cse_var_1 + 44)
- compute_4[cse_var_45] = (compute_4[cse_var_45] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_46: int32 = (cse_var_1 + 45)
- compute_4[cse_var_46] = (compute_4[cse_var_46] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_47: int32 = (cse_var_1 + 46)
- compute_4[cse_var_47] = (compute_4[cse_var_47] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_48: int32 = (cse_var_1 + 47)
- compute_4[cse_var_48] = (compute_4[cse_var_48] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 512)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_49: int32 = (cse_var_1 + 48)
- compute_4[cse_var_49] = (compute_4[cse_var_49] + (placeholder_16[((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16))]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_50: int32 = (cse_var_1 + 49)
- compute_4[cse_var_50] = (compute_4[cse_var_50] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 1)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_51: int32 = (cse_var_1 + 50)
- compute_4[cse_var_51] = (compute_4[cse_var_51] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 2)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_52: int32 = (cse_var_1 + 51)
- compute_4[cse_var_52] = (compute_4[cse_var_52] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 3)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_53: int32 = (cse_var_1 + 52)
- compute_4[cse_var_53] = (compute_4[cse_var_53] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 4)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_54: int32 = (cse_var_1 + 53)
- compute_4[cse_var_54] = (compute_4[cse_var_54] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 5)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_55: int32 = (cse_var_1 + 54)
- compute_4[cse_var_55] = (compute_4[cse_var_55] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 6)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_56: int32 = (cse_var_1 + 55)
- compute_4[cse_var_56] = (compute_4[cse_var_56] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 7)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_57: int32 = (cse_var_1 + 56)
- compute_4[cse_var_57] = (compute_4[cse_var_57] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 8)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_58: int32 = (cse_var_1 + 57)
- compute_4[cse_var_58] = (compute_4[cse_var_58] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 9)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_59: int32 = (cse_var_1 + 58)
- compute_4[cse_var_59] = (compute_4[cse_var_59] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 10)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_60: int32 = (cse_var_1 + 59)
- compute_4[cse_var_60] = (compute_4[cse_var_60] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 11)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_61: int32 = (cse_var_1 + 60)
- compute_4[cse_var_61] = (compute_4[cse_var_61] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 12)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_62: int32 = (cse_var_1 + 61)
- compute_4[cse_var_62] = (compute_4[cse_var_62] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 13)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_63: int32 = (cse_var_1 + 62)
- compute_4[cse_var_63] = (compute_4[cse_var_63] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 14)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
- }
- if @tir.likely((elem_idx < (placeholder_15[(i0.outer.i1.outer.fused + 1)] - placeholder_15[i0.outer.i1.outer.fused])), dtype=bool) {
- let cse_var_64: int32 = (cse_var_1 + 63)
- compute_4[cse_var_64] = (compute_4[cse_var_64] + (placeholder_16[(((placeholder_15[i0.outer.i1.outer.fused]*16) + (elem_idx*16)) + 15)]*max(placeholder_17[(((i.outer.inner*1024) + placeholder_18[(placeholder_15[i0.outer.i1.outer.fused] + elem_idx)]) + 768)], 0f32)))
+ for (i0.outer.i1.outer.fused: int32, 0, 1024) "parallel" {
+ allocate(compute_3: Pointer(global float32), float32, [64]), storage_scope = global {
+ for (i.inner.init: int32, 0, 4) {
+ for (j.init: int32, 0, 16) {
+ compute_4: Buffer(compute_3, float32, [64], [])[((i.inner.init*16) + j.init)] = 0f32
+ }
+ }
+ for (elem_idx: int32, 0, let cse_var_1: int32 = floormod(i0.outer.i1.outer.fused, 32) in (placeholder_15: Buffer(placeholder_13, int32, [33], [])[(cse_var_1 + 1)] - placeholder_15[cse_var_1])) {
+ for (i.inner: int32, 0, 4) {
+ for (j: int32, 0, 16) {
+ let cse_var_2: int32 = floormod(i0.outer.i1.outer.fused, 32)
+ if @tir.likely((elem_idx < (placeholder_15[(cse_var_2 + 1)] - placeholder_15[cse_var_2])), dtype=bool) {
+ let cse_var_3: int32 = ((i.inner*16) + j)
+ compute_4[cse_var_3] = (compute_4[cse_var_3] + (placeholder_16: Buffer(placeholder_11, float32, [78656], [])[(((placeholder_15[cse_var_2]*16) + (elem_idx*16)) + j)]*max(placeholder_17: Buffer(placeholder_10, float32, [32768], [])[(((floordiv(i0.outer.i1.outer.fused, 32)*1024) + (i.inner*256)) + placeholder_18: Buffer(placeholder_12, int32, [4916], [])[(placeholder_15[cse_var_2] + elem_idx)])], 0f32)))
}
}
}
}
- for (i0.inner: int32, 0, 128) {
- let cse_var_65: int32 = ((i0.inner*512) + (i0.outer.i1.outer.fused*16))
- compute_5: Buffer(compute_2, float32, [65536], [])[ramp(cse_var_65, 1, 16)] = max((compute_4[ramp((i0.inner*16), 1, 16)] + placeholder_19: Buffer(placeholder_14, float32, [65536], [])[ramp(cse_var_65, 1, 16)]), broadcast(0f32, 16))
+ for (i0.inner: int32, 0, 4) {
+ let cse_var_4: int32 = (((floordiv(i0.outer.i1.outer.fused, 32)*2048) + (i0.inner*512)) + (floormod(i0.outer.i1.outer.fused, 32)*16))
+ compute_5: Buffer(compute_2, float32, [65536], [])[ramp(cse_var_4, 1, 16)] = max((compute_4[ramp((i0.inner*16), 1, 16)] + placeholder_19: Buffer(placeholder_14, float32, [65536], [])[ramp(cse_var_4, 1, 16)]), broadcast(0f32, 16))
}
}
}
@@ -1000,7 +690,7 @@ layout transformation, parallelization, vectorization, unrolling, and operator f
<span class="p">)</span>
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Execution time of this operator: 3.211 ms
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Execution time of this operator: 1.333 ms
</pre></div>
</div>
<div class="admonition note">
diff --git a/docs/how_to/tune_with_autotvm/sg_execution_times.html b/docs/how_to/tune_with_autotvm/sg_execution_times.html
index c8391e563c..92ba40014e 100644
--- a/docs/how_to/tune_with_autotvm/sg_execution_times.html
+++ b/docs/how_to/tune_with_autotvm/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-tune-with-autotvm-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>00:26.988</strong> total execution time for <strong>how_to_tune_with_autotvm</strong> files:</p>
+<p><strong>00:26.542</strong> total execution time for <strong>how_to_tune_with_autotvm</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 84%" />
@@ -349,22 +349,22 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="tune_conv2d_cuda.html#sphx-glr-how-to-tune-with-autotvm-tune-conv2d-cuda-py"><span class="std std-ref">Tuning High Performance Convolution on NVIDIA GPUs</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_conv2d_cuda.py</span></code>)</p></td>
-<td><p>00:26.949</p></td>
+<td><p>00:26.507</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="tune_relay_x86.html#sphx-glr-how-to-tune-with-autotvm-tune-relay-x86-py"><span class="std std-ref">Auto-tuning a Convolutional Network for x86 CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_relay_x86.py</span></code>)</p></td>
-<td><p>00:00.024</p></td>
+<td><p>00:00.020</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="tune_relay_cuda.html#sphx-glr-how-to-tune-with-autotvm-tune-relay-cuda-py"><span class="std std-ref">Auto-tuning a Convolutional Network for NVIDIA GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_relay_cuda.py</span></code>)</p></td>
<td><p>00:00.005</p></td>
<td><p>0.0 MB</p></td>
</tr>
-<tr class="row-even"><td><p><a class="reference internal" href="tune_relay_arm.html#sphx-glr-how-to-tune-with-autotvm-tune-relay-arm-py"><span class="std std-ref">Auto-tuning a Convolutional Network for ARM CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_relay_arm.py</span></code>)</p></td>
+<tr class="row-even"><td><p><a class="reference internal" href="tune_relay_mobile_gpu.html#sphx-glr-how-to-tune-with-autotvm-tune-relay-mobile-gpu-py"><span class="std std-ref">Auto-tuning a Convolutional Network for Mobile GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_relay_mobile_gpu.py</span></code>)</p></td>
<td><p>00:00.005</p></td>
<td><p>0.0 MB</p></td>
</tr>
-<tr class="row-odd"><td><p><a class="reference internal" href="tune_relay_mobile_gpu.html#sphx-glr-how-to-tune-with-autotvm-tune-relay-mobile-gpu-py"><span class="std std-ref">Auto-tuning a Convolutional Network for Mobile GPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_relay_mobile_gpu.py</span></code>)</p></td>
+<tr class="row-odd"><td><p><a class="reference internal" href="tune_relay_arm.html#sphx-glr-how-to-tune-with-autotvm-tune-relay-arm-py"><span class="std std-ref">Auto-tuning a Convolutional Network for ARM CPU</span></a> (<code class="docutils literal notranslate"><span class="pre">tune_relay_arm.py</span></code>)</p></td>
<td><p>00:00.005</p></td>
<td><p>0.0 MB</p></td>
</tr>
diff --git a/docs/how_to/tune_with_autotvm/tune_conv2d_cuda.html b/docs/how_to/tune_with_autotvm/tune_conv2d_cuda.html
index a04c1b0df3..1b294fda1d 100644
--- a/docs/how_to/tune_with_autotvm/tune_conv2d_cuda.html
+++ b/docs/how_to/tune_with_autotvm/tune_conv2d_cuda.html
@@ -689,9 +689,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 1, 4]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 128, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1255863
-No: 2 GFLOPS: 103.32/103.32 result: MeasureResult(costs=(0.0022406029555555556,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.7429840564727783, timestamp=1670933521.5859842) [('tile_f', [-1, 1, 16, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,77914
-No: 3 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 4, 128, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,5600811
+No: 2 GFLOPS: 0.00/0.00 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -813,8 +812,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 8, 8]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 4, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,5851937
-No: 4 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 32, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 1, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,4877441
+No: 3 GFLOPS: 0.00/0.00 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -936,8 +935,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 4, 2]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 8]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,5140155
-No: 5 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 64, 2]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6378114
+No: 4 GFLOPS: 0.00/0.00 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1059,8 +1058,10 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 64, 2, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 16, 32]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1512956
-No: 6 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 64, 1, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 2, 32]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1500626
+No: 5 GFLOPS: 107.57/107.57 result: MeasureResult(costs=(0.002152012914893617,), error_no=MeasureErrorNo.NO_ERROR, all_cost=3.096332550048828, timestamp=1670949979.592225) [('tile_f', [-1, 2, 8, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9371368
+No: 6 GFLOPS: 47.50/107.57 result: MeasureResult(costs=(0.004874122619047619,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.0825843811035156, timestamp=1670949981.2732048) [('tile_f', [-1, 1, 32, 16]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 2, 1]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,586264
+No: 7 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1182,8 +1183,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 16, 1, 8]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 32, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,10122560
-No: 7 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 4, 32]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 16, 16]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,4975054
+No: 8 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1305,8 +1306,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 64, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6956666
-No: 8 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 2, 4]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 1, 256]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2120688
+No: 9 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1428,9 +1429,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 2, 128]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 128, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6064734
-No: 9 GFLOPS: 94.26/103.32 result: MeasureResult(costs=(0.0024560546829268293,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.2918369770050049, timestamp=1670933525.4022467) [('tile_f', [-1, 1, 64, 1]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 2, 32]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,146125
-No: 10 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 2, 16]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 4, 64]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,3459450
+No: 10 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1552,8 +1552,9 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 8, 2]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 16, 4]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,3955902
-No: 11 GFLOPS: 0.00/103.32 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 4, 32]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 8, 1]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,3304155
+No: 11 GFLOPS: 32.94/107.57 result: MeasureResult(costs=(0.007028339066666666,), error_no=MeasureErrorNo.NO_ERROR, all_cost=2.3546528816223145, timestamp=1670949984.7977166) [('tile_f', [-1, 1, 1, 2]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 32, 2]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,7992435
+No: 12 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1675,11 +1676,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 2, 32]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 2, 32]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9438633
-No: 12 GFLOPS: 27.72/103.32 result: MeasureResult(costs=(0.008351099105263158,), error_no=MeasureErrorNo.NO_ERROR, all_cost=1.592003583908081, timestamp=1670933526.43331) [('tile_f', [-1, 1, 1, 8]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 4, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1397576
-No: 13 GFLOPS: 9.08/103.32 result: MeasureResult(costs=(0.025502439249999998,), error_no=MeasureErrorNo.NO_ERROR, all_cost=2.2341725826263428, timestamp=1670933528.8338175) [('tile_f', [-1, 4, 8, 2]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 1]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2141781
-No: 14 GFLOPS: 132.18/132.18 result: MeasureResult(costs=(0.0017514406195652176,), error_no=MeasureErrorNo.NO_ERROR, all_cost=2.2057106494903564, timestamp=1670933529.8254948) [('tile_f', [-1, 16, 4, 2]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 16, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,3535916
-No: 15 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 128, 4, 1]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 128, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,643086
+No: 13 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1801,8 +1799,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 8, 64, 1]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 32, 1]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9698968
-No: 16 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 1, 128]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 32, 8]), ('tile_ry', [-1, 1, 3]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,8633011
+No: 14 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -1924,8 +1922,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 4, 32, 2]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 8, 1]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2529432
-No: 17 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 32, 4, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 2, 256]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1351044
+No: 15 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -2047,8 +2045,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 32, 4]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 8, 32]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,7316451
-No: 18 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 128, 1, 4]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 1, 512]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6774567
+No: 16 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -2170,8 +2168,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 16, 2, 1]), ('tile_y', [-1, 1, 1, 7]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 4, 128]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1341794
-No: 19 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 128, 2]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 256, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 512), ('unroll_explicit', 0)],None,2387978
+No: 17 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -2293,8 +2291,8 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 32, 16, 1]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 16, 1]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 512), ('unroll_explicit', 1)],None,8338919
-No: 20 GFLOPS: 0.00/132.18 result: Traceback (most recent call last):
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 4, 1, 16]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 8, 32]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,10025566
+No: 18 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
@@ -2416,7 +2414,253 @@ Traceback (most recent call last):
File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
raise InstantiationError("Skipped because of invalid gpu kernel")
-tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 2, 4, 8]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 1, 256]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,957590
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 512, 1]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 1, 7]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 0), ('unroll_explicit', 1)],None,6081734
+No: 19 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
+ func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
+ func = build(s, args, target_host=task.target_host, runtime=runtime)
+ File "/workspace/python/tvm/driver/build_module.py", line 227, in build
+ input_mod = lower(inputs, args, name=name, binds=binds)
+ File "/workspace/python/tvm/driver/build_module.py", line 134, in lower
+ return ffi.lower_schedule(inp, args, name, binds, simple_mode)
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 331, in tvm._ffi._cy3.core.PackedFuncBase.__call__
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 276, in tvm._ffi._cy3.core.FuncCall
+ File "tvm/_ffi/_cython/./base.pxi", line 181, in tvm._ffi._cy3.core.CHECK_CALL
+tvm._ffi.base.TVMError: Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel
+
+Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 512, 1, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 256, 2]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 3]), ('auto_unroll_max_step', 0), ('unroll_explicit', 0)],None,1419669
+No: 20 GFLOPS: 0.00/107.57 result: Traceback (most recent call last):
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 592, in __call__
+ func, arg_info = _build_func_common(measure_input, self.runtime, **kwargs)
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 544, in _build_func_common
+ func = build(s, args, target_host=task.target_host, runtime=runtime)
+ File "/workspace/python/tvm/driver/build_module.py", line 227, in build
+ input_mod = lower(inputs, args, name=name, binds=binds)
+ File "/workspace/python/tvm/driver/build_module.py", line 134, in lower
+ return ffi.lower_schedule(inp, args, name, binds, simple_mode)
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 331, in tvm._ffi._cy3.core.PackedFuncBase.__call__
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 276, in tvm._ffi._cy3.core.FuncCall
+ File "tvm/_ffi/_cython/./base.pxi", line 181, in tvm._ffi._cy3.core.CHECK_CALL
+tvm._ffi.base.TVMError: Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel
+
+Traceback (most recent call last):
+ 24: TVMFuncCall
+ at ../src/runtime/c_runtime_api.cc:477
+ 23: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 22: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 21: operator()
+ at ../include/tvm/runtime/packed_func.h:1730
+ 20: unpack_call<tvm::IRModule, 5, tvm::<lambda(tvm::te::Schedule, const tvm::runtime::Array<tvm::runtime::ObjectRef>&, const tvm::runtime::String&, const tvm::runtime::Map<tvm::te::Tensor, tvm::tir::Buffer>&, bool)> >
+ at ../include/tvm/runtime/packed_func.h:1670
+ 19: run<>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 18: run<tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 17: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 16: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 15: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1630
+ 14: run<tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_, tvm::runtime::TVMMovableArgValueWithContext_>
+ at ../include/tvm/runtime/packed_func.h:1645
+ 13: operator()
+ at ../src/driver/driver_api.cc:388
+ 12: tvm::LowerSchedule(tvm::te::Schedule, tvm::runtime::Array<tvm::runtime::ObjectRef, void> const&, std::__cxx11::basic_string<char, std::char_traits<char>, std::allocator<char> > const&, std::unordered_map<tvm::te::Tensor, tvm::tir::Buffer, std::hash<tvm::te::Tensor>, std::equal_to<tvm::te::Tensor>, std::allocator<std::pair<tvm::te::Tensor const, tvm::tir::Buffer> > > const&, tvm::GlobalVarSupply, bool)
+ at ../src/driver/driver_api.cc:374
+ 11: tvm::LowerWithPassList(tvm::IRModule, tvm::runtime::Array<tvm::transform::Pass, void>)
+ at ../src/driver/driver_api.cc:269
+ 10: tvm::transform::Pass::operator()(tvm::IRModule) const
+ at ../src/ir/transform.cc:258
+ 9: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 8: tvm::transform::SequentialNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:453
+ 7: tvm::transform::Pass::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/ir/transform.cc:274
+ 6: tvm::tir::transform::PrimFuncPassNode::operator()(tvm::IRModule, tvm::transform::PassContext const&) const
+ at ../src/tir/ir/transform.cc:100
+ 5: tvm::runtime::TypedPackedFunc<tvm::tir::PrimFunc (tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext)>::operator()(tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext) const
+ at ../include/tvm/runtime/packed_func.h:1749
+ 4: tvm::tir::PrimFunc tvm::runtime::detail::typed_packed_call_dispatcher<tvm::tir::PrimFunc>::run<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::runtime::PackedFunc const&, tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&)
+ at ../include/tvm/runtime/packed_func.h:1693
+ 3: tvm::runtime::TVMRetValue tvm::runtime::PackedFunc::operator()<tvm::tir::PrimFunc, tvm::IRModule, tvm::transform::PassContext>(tvm::tir::PrimFunc&&, tvm::IRModule&&, tvm::transform::PassContext&&) const
+ at ../include/tvm/runtime/packed_func.h:1617
+ 2: tvm::runtime::PackedFuncObj::CallPacked(tvm::runtime::TVMArgs, tvm::runtime::TVMRetValue*) const
+ at ../include/tvm/runtime/packed_func.h:1217
+ 1: Call
+ at ../include/tvm/runtime/packed_func.h:1213
+ 0: operator()
+ at ../src/runtime/c_runtime_api.cc:534
+ File "tvm/_ffi/_cython/./packed_func.pxi", line 56, in tvm._ffi._cy3.core.tvm_callback
+ File "/workspace/python/tvm/autotvm/measure/measure_methods.py", line 875, in verify_pass
+ raise InstantiationError("Skipped because of invalid gpu kernel")
+tvm.autotvm.task.space.InstantiationError: Skipped because of invalid gpu kernel [('tile_f', [-1, 1, 16, 8]), ('tile_y', [-1, 1, 7, 1]), ('tile_x', [-1, 1, 1, 1]), ('tile_rc', [-1, 32, 16]), ('tile_ry', [-1, 3, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9043478
</pre></div>
</div>
<p>Finally we can inspect the best config from log file, check correctness,
@@ -2455,9 +2699,9 @@ and measure running time.</p>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Finish loading 20 records
Best config:
-[('tile_f', [-1, 16, 4, 2]), ('tile_y', [-1, 1, 1, 1]), ('tile_x', [-1, 1, 7, 1]), ('tile_rc', [-1, 16, 2]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 1, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 0)],None,3535916
+[('tile_f', [-1, 2, 8, 1]), ('tile_y', [-1, 7, 1, 1]), ('tile_x', [-1, 7, 1, 1]), ('tile_rc', [-1, 8, 4]), ('tile_ry', [-1, 1, 1]), ('tile_rx', [-1, 3, 1]), ('auto_unroll_max_step', 1500), ('unroll_explicit', 1)],None,9371368
Finish loading 20 records
-Time cost of this operator: 0.002177
+Time cost of this operator: 0.001173
</pre></div>
</div>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-tune-with-autotvm-tune-conv2d-cuda-py">
diff --git a/docs/how_to/work_with_microtvm/micro_autotune.html b/docs/how_to/work_with_microtvm/micro_autotune.html
index 944c8fcef6..42de336e3b 100644
--- a/docs/how_to/work_with_microtvm/micro_autotune.html
+++ b/docs/how_to/work_with_microtvm/micro_autotune.html
@@ -598,10 +598,10 @@ the tuned operator.</p>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>########## Build without Autotuning ##########
Node Name Ops Time(us) Time(%) Shape Inputs Outputs Measurements(us)
--------- --- -------- ------- ----- ------ ------- ----------------
-tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 310.3 98.636 (1, 2, 10, 10, 3) 2 1 [310.3]
-tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 3.164 1.006 (1, 6, 10, 10) 1 1 [3.164]
-tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 1.127 0.358 (1, 1, 10, 10, 3) 1 1 [1.127]
-Total_time - 314.592 - - - - -
+tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 311.0 98.728 (1, 2, 10, 10, 3) 2 1 [311.0]
+tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 3.036 0.964 (1, 6, 10, 10) 1 1 [3.036]
+tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 0.97 0.308 (1, 1, 10, 10, 3) 1 1 [0.97]
+Total_time - 315.006 - - - - -
</pre></div>
</div>
</div>
@@ -653,10 +653,10 @@ Total_time -
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>########## Build with Autotuning ##########
Node Name Ops Time(us) Time(%) Shape Inputs Outputs Measurements(us)
--------- --- -------- ------- ----- ------ ------- ----------------
-tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 105.0 97.55 (1, 6, 10, 10, 1) 2 1 [105.0]
-tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 1.792 1.665 (1, 6, 10, 10) 1 1 [1.792]
-tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 0.845 0.785 (1, 3, 10, 10, 1) 1 1 [0.845]
-Total_time - 107.637 - - - - -
+tvmgen_default_fused_nn_contrib_conv2d_NCHWc tvmgen_default_fused_nn_contrib_conv2d_NCHWc 100.1 97.276 (1, 6, 10, 10, 1) 2 1 [100.1]
+tvmgen_default_fused_layout_transform_1 tvmgen_default_fused_layout_transform_1 1.78 1.729 (1, 6, 10, 10) 1 1 [1.78]
+tvmgen_default_fused_layout_transform tvmgen_default_fused_layout_transform 1.023 0.994 (1, 1, 10, 10, 3) 1 1 [1.023]
+Total_time - 102.903 - - - - -
</pre></div>
</div>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-work-with-microtvm-micro-autotune-py">
diff --git a/docs/how_to/work_with_microtvm/micro_pytorch.html b/docs/how_to/work_with_microtvm/micro_pytorch.html
index ddbb33de9d..9866a44bdf 100644
--- a/docs/how_to/work_with_microtvm/micro_pytorch.html
+++ b/docs/how_to/work_with_microtvm/micro_pytorch.html
@@ -440,7 +440,8 @@ download a cat image and preprocess it to use as the model input.</p>
Downloading: "https://download.pytorch.org/models/quantized/mobilenet_v2_qnnpack_37f702c5.pth" to /workspace/.cache/torch/hub/checkpoints/mobilenet_v2_qnnpack_37f702c5.pth
0%| | 0.00/3.42M [00:00<?, ?B/s]
-100%|##########| 3.42M/3.42M [00:00<00:00, 61.6MB/s]
+ 70%|######9 | 2.39M/3.42M [00:00<00:00, 25.1MB/s]
+100%|##########| 3.42M/3.42M [00:00<00:00, 34.3MB/s]
/workspace/python/tvm/relay/frontend/pytorch_utils.py:47: DeprecationWarning: distutils Version classes are deprecated. Use packaging.version instead.
return LooseVersion(torch_ver) > ver
/venv/apache-tvm-py3.7/lib/python3.7/site-packages/setuptools/_distutils/version.py:346: DeprecationWarning: distutils Version classes are deprecated. Use packaging.version instead.
@@ -564,7 +565,7 @@ via the host <cite>main.cc`</cite> or if a Zephyr emulated board is selected as
Torch top-1 id: 282, class name: tiger cat
</pre></div>
</div>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 3.019 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 1 minutes 2.998 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-work-with-microtvm-micro-pytorch-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/12b9ecc04c41abaa12022061771821d1/micro_pytorch.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">micro_pytorch.py</span></code></a></p>
diff --git a/docs/how_to/work_with_microtvm/micro_train.html b/docs/how_to/work_with_microtvm/micro_train.html
index 1779eccd1f..a7e1bfc3dd 100644
--- a/docs/how_to/work_with_microtvm/micro_train.html
+++ b/docs/how_to/work_with_microtvm/micro_train.html
@@ -530,7 +530,7 @@ take about <strong>2 minutes</strong> to download the Stanford Cars, while COCO
<a href="https://docs.python.org/3/library/shutil.html#shutil.move" title="shutil.move" class="sphx-glr-backref-module-shutil sphx-glr-backref-type-py-function"><span class="n">shutil</span><span class="o">.</span><span class="n">move</span></a><span class="p">(</span><span class="sa">f</span><span class="s2">"</span><span class="si">{</span><a href="https://docs.python.org/3/library/stdtypes.html#str" title="builtins.str" class="sphx-glr-backref-module-builtins sphx-glr-backref-typ [...]
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>'/tmp/tmp_8kq7sx6/images/random'
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>'/tmp/tmpn64xoprz/images/random'
</pre></div>
</div>
</div>
@@ -590,8 +590,8 @@ objects to other stuff? We can display some examples from our datasets using <co
<span class="n">plt</span><span class="o">.</span><span class="n">axis</span><span class="p">(</span><span class="s2">"off"</span><span class="p">)</span>
</pre></div>
</div>
-<img src="../../_images/sphx_glr_micro_train_001.png" srcset="../../_images/sphx_glr_micro_train_001.png" alt="[0.0, 1.0], [0.0, 1.0], [1.0, 0.0], [1.0, 0.0], [1.0, 0.0], [1.0, 0.0], [0.0, 1.0], [0.0, 1.0], [0.0, 1.0], [1.0, 0.0]" class = "sphx-glr-single-img"/><div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>/tmp/tmp_8kq7sx6/images/target contains 8144 images
-/tmp/tmp_8kq7sx6/images/random contains 5000 images
+<img src="../../_images/sphx_glr_micro_train_001.png" srcset="../../_images/sphx_glr_micro_train_001.png" alt="[1.0, 0.0], [0.0, 1.0], [0.0, 1.0], [1.0, 0.0], [1.0, 0.0], [0.0, 1.0], [1.0, 0.0], [1.0, 0.0], [1.0, 0.0], [0.0, 1.0]" class = "sphx-glr-single-img"/><div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>/tmp/tmpn64xoprz/images/target contains 8144 images
+/tmp/tmpn64xoprz/images/random contains 5000 images
</pre></div>
</div>
</div>
@@ -703,13 +703,13 @@ the time on our validation set).</p>
</pre></div>
</div>
<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span>Epoch 1/3
-328/328 - 47s - loss: 0.2257 - accuracy: 0.9235 - val_loss: 0.2299 - val_accuracy: 0.9313 - 47s/epoch - 142ms/step
+328/328 - 47s - loss: 0.2277 - accuracy: 0.9188 - val_loss: 0.1379 - val_accuracy: 0.9535 - 47s/epoch - 144ms/step
Epoch 2/3
-328/328 - 43s - loss: 0.0990 - accuracy: 0.9648 - val_loss: 0.1212 - val_accuracy: 0.9619 - 43s/epoch - 132ms/step
+328/328 - 43s - loss: 0.0917 - accuracy: 0.9676 - val_loss: 0.1415 - val_accuracy: 0.9573 - 43s/epoch - 132ms/step
Epoch 3/3
-328/328 - 43s - loss: 0.0623 - accuracy: 0.9754 - val_loss: 0.1194 - val_accuracy: 0.9698 - 43s/epoch - 131ms/step
+328/328 - 43s - loss: 0.0686 - accuracy: 0.9754 - val_loss: 0.1399 - val_accuracy: 0.9607 - 43s/epoch - 132ms/step
-<keras.callbacks.History object at 0x7ff68c133210>
+<keras.callbacks.History object at 0x7f8616e6cb50>
</pre></div>
</div>
</div>
@@ -971,7 +971,7 @@ as intended.</p>
<p>From here, we could modify the model to read live images from the camera - we have another
Arduino tutorial for how to do that <a class="reference external" href="https://github.com/guberti/tvm-arduino-demos/tree/master/examples/person_detection">on GitHub</a>. Alternatively, we could also
<a class="reference external" href="https://tvm.apache.org/docs/how_to/work_with_microtvm/micro_autotune.html">use TVM’s autotuning capabilities</a> to dramatically improve the model’s performance.</p>
-<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 5 minutes 21.287 seconds)</p>
+<p class="sphx-glr-timing"><strong>Total running time of the script:</strong> ( 4 minutes 34.133 seconds)</p>
<div class="sphx-glr-footer sphx-glr-footer-example docutils container" id="sphx-glr-download-how-to-work-with-microtvm-micro-train-py">
<div class="sphx-glr-download sphx-glr-download-python docutils container">
<p><a class="reference download internal" download="" href="../../_downloads/b52cec46baf4f78d6bcd94cbe269c8a6/micro_train.py"><code class="xref download docutils literal notranslate"><span class="pre">Download</span> <span class="pre">Python</span> <span class="pre">source</span> <span class="pre">code:</span> <span class="pre">micro_train.py</span></code></a></p>
diff --git a/docs/how_to/work_with_microtvm/sg_execution_times.html b/docs/how_to/work_with_microtvm/sg_execution_times.html
index eefa1ece8d..fd3f6f468b 100644
--- a/docs/how_to/work_with_microtvm/sg_execution_times.html
+++ b/docs/how_to/work_with_microtvm/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-work-with-microtvm-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>07:26.613</strong> total execution time for <strong>how_to_work_with_microtvm</strong> files:</p>
+<p><strong>06:40.496</strong> total execution time for <strong>how_to_work_with_microtvm</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 83%" />
@@ -349,23 +349,23 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="micro_train.html#sphx-glr-how-to-work-with-microtvm-micro-train-py"><span class="std std-ref">Training Vision Models for microTVM on Arduino</span></a> (<code class="docutils literal notranslate"><span class="pre">micro_train.py</span></code>)</p></td>
-<td><p>05:21.287</p></td>
+<td><p>04:34.133</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="micro_pytorch.html#sphx-glr-how-to-work-with-microtvm-micro-pytorch-py"><span class="std std-ref">microTVM PyTorch Tutorial</span></a> (<code class="docutils literal notranslate"><span class="pre">micro_pytorch.py</span></code>)</p></td>
-<td><p>01:03.019</p></td>
+<td><p>01:02.998</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="micro_autotune.html#sphx-glr-how-to-work-with-microtvm-micro-autotune-py"><span class="std std-ref">Autotuning with microTVM</span></a> (<code class="docutils literal notranslate"><span class="pre">micro_autotune.py</span></code>)</p></td>
-<td><p>00:50.848</p></td>
+<td><p>00:51.620</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="micro_aot.html#sphx-glr-how-to-work-with-microtvm-micro-aot-py"><span class="std std-ref">microTVM Host-Driven AoT</span></a> (<code class="docutils literal notranslate"><span class="pre">micro_aot.py</span></code>)</p></td>
-<td><p>00:07.681</p></td>
+<td><p>00:07.914</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="micro_tflite.html#sphx-glr-how-to-work-with-microtvm-micro-tflite-py"><span class="std std-ref">microTVM with TFLite Models</span></a> (<code class="docutils literal notranslate"><span class="pre">micro_tflite.py</span></code>)</p></td>
-<td><p>00:03.777</p></td>
+<td><p>00:03.829</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="micro_reference_vm.html#sphx-glr-how-to-work-with-microtvm-micro-reference-vm-py"><span class="std std-ref">microTVM Reference Virtual Machines</span></a> (<code class="docutils literal notranslate"><span class="pre">micro_reference_vm.py</span></code>)</p></td>
diff --git a/docs/how_to/work_with_relay/sg_execution_times.html b/docs/how_to/work_with_relay/sg_execution_times.html
index 691b103e70..9401a6c7c0 100644
--- a/docs/how_to/work_with_relay/sg_execution_times.html
+++ b/docs/how_to/work_with_relay/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-work-with-relay-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>00:44.009</strong> total execution time for <strong>how_to_work_with_relay</strong> files:</p>
+<p><strong>00:44.406</strong> total execution time for <strong>how_to_work_with_relay</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 84%" />
@@ -349,15 +349,15 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="using_pipeline_executor.html#sphx-glr-how-to-work-with-relay-using-pipeline-executor-py"><span class="std std-ref">Using Pipeline Executor in Relay</span></a> (<code class="docutils literal notranslate"><span class="pre">using_pipeline_executor.py</span></code>)</p></td>
-<td><p>00:32.270</p></td>
+<td><p>00:32.573</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="using_external_lib.html#sphx-glr-how-to-work-with-relay-using-external-lib-py"><span class="std std-ref">Using External Libraries in Relay</span></a> (<code class="docutils literal notranslate"><span class="pre">using_external_lib.py</span></code>)</p></td>
-<td><p>00:10.201</p></td>
+<td><p>00:10.261</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="build_gcn.html#sphx-glr-how-to-work-with-relay-build-gcn-py"><span class="std std-ref">Building a Graph Convolutional Network</span></a> (<code class="docutils literal notranslate"><span class="pre">build_gcn.py</span></code>)</p></td>
-<td><p>00:01.531</p></td>
+<td><p>00:01.565</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="using_relay_viz.html#sphx-glr-how-to-work-with-relay-using-relay-viz-py"><span class="std std-ref">Use Relay Visualizer to Visualize Relay</span></a> (<code class="docutils literal notranslate"><span class="pre">using_relay_viz.py</span></code>)</p></td>
diff --git a/docs/how_to/work_with_schedules/intrin_math.html b/docs/how_to/work_with_schedules/intrin_math.html
index 78b5829d66..5eeef71077 100644
--- a/docs/how_to/work_with_schedules/intrin_math.html
+++ b/docs/how_to/work_with_schedules/intrin_math.html
@@ -535,7 +535,7 @@ The following example customizes CUDA lowering rule for <code class="code docuti
<a href="../../reference/api/python/ir.html#tvm.ir.register_intrin_lowering" title="tvm.ir.register_intrin_lowering" class="sphx-glr-backref-module-tvm-ir sphx-glr-backref-type-py-function"><span class="n">register_intrin_lowering</span></a><span class="p">(</span><span class="s2">"tir.exp"</span><span class="p">,</span> <span class="n">target</span><span class="o">=</span><span class="s2">"cuda"</span><span class="p">,</span> <span class="n">f</span><span class="o">= [...]
</pre></div>
</div>
-<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span><function my_cuda_math_rule at 0x7ff67e92a320>
+<div class="sphx-glr-script-out highlight-none notranslate"><div class="highlight"><pre><span></span><function my_cuda_math_rule at 0x7f86122ac170>
</pre></div>
</div>
<p>Register the rule to TVM with override option to override existing rule.
diff --git a/docs/how_to/work_with_schedules/sg_execution_times.html b/docs/how_to/work_with_schedules/sg_execution_times.html
index 2cdd2fbefa..06c89b2bcc 100644
--- a/docs/how_to/work_with_schedules/sg_execution_times.html
+++ b/docs/how_to/work_with_schedules/sg_execution_times.html
@@ -340,7 +340,7 @@
<div class="section" id="computation-times">
<span id="sphx-glr-how-to-work-with-schedules-sg-execution-times"></span><h1>Computation times<a class="headerlink" href="#computation-times" title="Permalink to this headline">¶</a></h1>
-<p><strong>00:06.392</strong> total execution time for <strong>how_to_work_with_schedules</strong> files:</p>
+<p><strong>00:08.291</strong> total execution time for <strong>how_to_work_with_schedules</strong> files:</p>
<table class="docutils align-default">
<colgroup>
<col style="width: 83%" />
@@ -349,15 +349,15 @@
</colgroup>
<tbody>
<tr class="row-odd"><td><p><a class="reference internal" href="intrin_math.html#sphx-glr-how-to-work-with-schedules-intrin-math-py"><span class="std std-ref">Intrinsics and Math Functions</span></a> (<code class="docutils literal notranslate"><span class="pre">intrin_math.py</span></code>)</p></td>
-<td><p>00:03.827</p></td>
+<td><p>00:05.782</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="tensorize.html#sphx-glr-how-to-work-with-schedules-tensorize-py"><span class="std std-ref">Use Tensorize to Leverage Hardware Intrinsics</span></a> (<code class="docutils literal notranslate"><span class="pre">tensorize.py</span></code>)</p></td>
-<td><p>00:01.214</p></td>
+<td><p>00:01.153</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="reduction.html#sphx-glr-how-to-work-with-schedules-reduction-py"><span class="std std-ref">Reduction</span></a> (<code class="docutils literal notranslate"><span class="pre">reduction.py</span></code>)</p></td>
-<td><p>00:00.578</p></td>
+<td><p>00:00.579</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="scan.html#sphx-glr-how-to-work-with-schedules-scan-py"><span class="std std-ref">Scan and Recurrent Kernel</span></a> (<code class="docutils literal notranslate"><span class="pre">scan.py</span></code>)</p></td>
@@ -365,19 +365,19 @@
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="extern_op.html#sphx-glr-how-to-work-with-schedules-extern-op-py"><span class="std std-ref">External Tensor Functions</span></a> (<code class="docutils literal notranslate"><span class="pre">extern_op.py</span></code>)</p></td>
-<td><p>00:00.114</p></td>
+<td><p>00:00.115</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="schedule_primitives.html#sphx-glr-how-to-work-with-schedules-schedule-primitives-py"><span class="std std-ref">Schedule Primitives in TVM</span></a> (<code class="docutils literal notranslate"><span class="pre">schedule_primitives.py</span></code>)</p></td>
-<td><p>00:00.050</p></td>
+<td><p>00:00.052</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-odd"><td><p><a class="reference internal" href="tedd.html#sphx-glr-how-to-work-with-schedules-tedd-py"><span class="std std-ref">Use Tensor Expression Debug Display (TEDD) for Visualization</span></a> (<code class="docutils literal notranslate"><span class="pre">tedd.py</span></code>)</p></td>
-<td><p>00:00.029</p></td>
+<td><p>00:00.030</p></td>
<td><p>0.0 MB</p></td>
</tr>
<tr class="row-even"><td><p><a class="reference internal" href="tuple_inputs.html#sphx-glr-how-to-work-with-schedules-tuple-inputs-py"><span class="std std-ref">Compute and Reduce with Tuple Inputs</span></a> (<code class="docutils literal notranslate"><span class="pre">tuple_inputs.py</span></code>)</p></td>
-<td><p>00:00.023</p></td>
+<td><p>00:00.025</p></td>
<td><p>0.0 MB</p></td>
</tr>
</tbody>
diff --git a/docs/how_to/work_with_schedules/tensorize.html b/docs/how_to/work_with_schedules/tensorize.html
index d372649469..1e98ffc769 100644
--- a/docs/how_to/work_with_schedules/tensorize.html
+++ b/docs/how_to/work_with_schedules/tensorize.html
@@ -586,7 +586,7 @@ The importing needs to happen before the tensorized GEMV being executed.</p>
B: Buffer(B_2: Pointer(float32), float32, [512, 64], []),
C: Buffer(C_2: Pointer(float32), float32, [1024, 512], [])}
buffer_map = {A_1: A, B_1: B, C_1: C} {
- attr [IterVar(i: int32, (nullptr), "DataPar", "")] "pragma_import_llvm" = "; ModuleID = '/tmp/tmpcbhkriab/input0.cc'\nsource_filename = \"/tmp/tmpcbhkriab/input0.cc\"\ntarget datalayout = \"e-m:e-i64:64-f80:128-n8:16:32:64-S128\"\ntarget triple = \"x86_64-pc-linux-gnu\"\n\n; Function Attrs: noinline nounwind optnone uwtable\ndefine dso_local i32 @gemv_update(float*, float*, float*, i32, i32, i32) #0 {\n %7 = allo [...]
+ attr [IterVar(i: int32, (nullptr), "DataPar", "")] "pragma_import_llvm" = "; ModuleID = '/tmp/tmp86lnfzzk/input0.cc'\nsource_filename = \"/tmp/tmp86lnfzzk/input0.cc\"\ntarget datalayout = \"e-m:e-i64:64-f80:128-n8:16:32:64-S128\"\ntarget triple = \"x86_64-pc-linux-gnu\"\n\n; Function Attrs: noinline nounwind optnone uwtable\ndefine dso_local i32 @gemv_update(float*, float*, float*, i32, i32, i32) #0 {\n %7 = allo [...]
for (i, 0, 1024) {
for (j.outer: int32, 0, 32) {
@tir.call_extern("gemv_update", @tir.tvm_access_ptr(@tir.type_annotation(, dtype=float32), C_2, ((i*512) + (j.outer*16)), 16, 2, dtype=handle), @tir.tvm_access_ptr(@tir.type_annotation(, dtype=float32), A_2, (i*64), 64, 1, dtype=handle), @tir.tvm_access_ptr(@tir.type_annotation(, dtype=float32), B_2, (j.outer*1024), 1024, 1, dtype=handle), 16, 64, 64, dtype=int32)
diff --git a/docs/reference/api/python/auto_scheduler.html b/docs/reference/api/python/auto_scheduler.html
index c8c9d54a22..ee5a2dad9d 100644
--- a/docs/reference/api/python/auto_scheduler.html
+++ b/docs/reference/api/python/auto_scheduler.html
@@ -1615,7 +1615,7 @@ history states as starting point to perform Evolutionary Search).</p></li>
... 2970 lines suppressed ...