初始化项目,由ModelHub XC社区提供模型
Model: aws-neuron/Qwen3-1.7B-TP4-BS4-SEQ2048 Source: Original Platform
This commit is contained in:
59
.gitattributes
vendored
Normal file
59
.gitattributes
vendored
Normal file
@@ -0,0 +1,59 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk0/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk0/model.MODULE_bdcbea8455ebae357b4c+541d7181.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk1/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk1/model.MODULE_e7dad336ed1c266a3016+bbc3fa47.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk2/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk2/model.MODULE_f36c9ad51e28c98c9723+c4081b94.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk3/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk3/model.MODULE_0984f4c19a044cc11c2a+1c024d2c.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk4/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
context_encoding_model/_tp0_bk4/model.MODULE_e6b44ff520e6c4333666+e9aa1481.neff filter=lfs diff=lfs merge=lfs -text
|
||||
layout_opt/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
layout_opt/model/graph.hlo filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk0/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk0/model.MODULE_5e3d0ce8963512f9ea2d+937cd7a2.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk0/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk1/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk1/model.MODULE_f53407701fa4882a24c0+55c11e15.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk2/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk2/model.MODULE_c60e20a111715e620aa2+6bb8acc9.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk3/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk3/model.MODULE_02e08d5fdadfbcb8a569+a6f96e4f.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk4/graph.neff filter=lfs diff=lfs merge=lfs -text
|
||||
token_generation_model/_tp0_bk4/model.MODULE_b4b44a47076a167bd9a5+795ee7cc.neff filter=lfs diff=lfs merge=lfs -text
|
||||
202
LICENSE
Normal file
202
LICENSE
Normal file
@@ -0,0 +1,202 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright 2024 Alibaba Cloud
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
301
README.md
Normal file
301
README.md
Normal file
@@ -0,0 +1,301 @@
|
||||
---
|
||||
library_name: transformers
|
||||
license: apache-2.0
|
||||
license_link: https://huggingface.co/Qwen/Qwen3-1.7B/blob/main/LICENSE
|
||||
pipeline_tag: text-generation
|
||||
base_model:
|
||||
- Qwen/Qwen3-1.7B-Base
|
||||
---
|
||||
|
||||
# Qwen3-1.7B
|
||||
<a href="https://chat.qwen.ai/" target="_blank" style="margin: 2px;">
|
||||
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
|
||||
</a>
|
||||
|
||||
## Qwen3 Highlights
|
||||
|
||||
Qwen3 is the latest generation of large language models in Qwen series, offering a comprehensive suite of dense and mixture-of-experts (MoE) models. Built upon extensive training, Qwen3 delivers groundbreaking advancements in reasoning, instruction-following, agent capabilities, and multilingual support, with the following key features:
|
||||
|
||||
- **Uniquely support of seamless switching between thinking mode** (for complex logical reasoning, math, and coding) and **non-thinking mode** (for efficient, general-purpose dialogue) **within single model**, ensuring optimal performance across various scenarios.
|
||||
- **Significantly enhancement in its reasoning capabilities**, surpassing previous QwQ (in thinking mode) and Qwen2.5 instruct models (in non-thinking mode) on mathematics, code generation, and commonsense logical reasoning.
|
||||
- **Superior human preference alignment**, excelling in creative writing, role-playing, multi-turn dialogues, and instruction following, to deliver a more natural, engaging, and immersive conversational experience.
|
||||
- **Expertise in agent capabilities**, enabling precise integration with external tools in both thinking and unthinking modes and achieving leading performance among open-source models in complex agent-based tasks.
|
||||
- **Support of 100+ languages and dialects** with strong capabilities for **multilingual instruction following** and **translation**.
|
||||
|
||||
## Model Overview
|
||||
|
||||
**Qwen3-1.7B** has the following features:
|
||||
- Type: Causal Language Models
|
||||
- Training Stage: Pretraining & Post-training
|
||||
- Number of Parameters: 1.7B
|
||||
- Number of Paramaters (Non-Embedding): 1.4B
|
||||
- Number of Layers: 28
|
||||
- Number of Attention Heads (GQA): 16 for Q and 8 for KV
|
||||
- Context Length: 32,768
|
||||
|
||||
For more details, including benchmark evaluation, hardware requirements, and inference performance, please refer to our [blog](https://qwenlm.github.io/blog/qwen3/), [GitHub](https://github.com/QwenLM/Qwen3), and [Documentation](https://qwen.readthedocs.io/en/latest/).
|
||||
|
||||
> [!TIP]
|
||||
> If you encounter significant endless repetitions, please refer to the [Best Practices](#best-practices) section for optimal sampling parameters, and set the ``presence_penalty`` to 1.5.
|
||||
|
||||
## Quickstart
|
||||
|
||||
The code of Qwen3 has been in the latest Hugging Face `transformers` and we advise you to use the latest version of `transformers`.
|
||||
|
||||
With `transformers<4.51.0`, you will encounter the following error:
|
||||
```
|
||||
KeyError: 'qwen3'
|
||||
```
|
||||
|
||||
The following contains a code snippet illustrating how to use the model generate content based on given inputs.
|
||||
```python
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
model_name = "Qwen/Qwen3-1.7B"
|
||||
|
||||
# load the tokenizer and the model
|
||||
tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
model_name,
|
||||
torch_dtype="auto",
|
||||
device_map="auto"
|
||||
)
|
||||
|
||||
# prepare the model input
|
||||
prompt = "Give me a short introduction to large language model."
|
||||
messages = [
|
||||
{"role": "user", "content": prompt}
|
||||
]
|
||||
text = tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
enable_thinking=True # Switches between thinking and non-thinking modes. Default is True.
|
||||
)
|
||||
model_inputs = tokenizer([text], return_tensors="pt").to(model.device)
|
||||
|
||||
# conduct text completion
|
||||
generated_ids = model.generate(
|
||||
**model_inputs,
|
||||
max_new_tokens=32768
|
||||
)
|
||||
output_ids = generated_ids[0][len(model_inputs.input_ids[0]):].tolist()
|
||||
|
||||
# parsing thinking content
|
||||
try:
|
||||
# rindex finding 151668 (</think>)
|
||||
index = len(output_ids) - output_ids[::-1].index(151668)
|
||||
except ValueError:
|
||||
index = 0
|
||||
|
||||
thinking_content = tokenizer.decode(output_ids[:index], skip_special_tokens=True).strip("\n")
|
||||
content = tokenizer.decode(output_ids[index:], skip_special_tokens=True).strip("\n")
|
||||
|
||||
print("thinking content:", thinking_content)
|
||||
print("content:", content)
|
||||
```
|
||||
|
||||
For deployment, you can use `sglang>=0.4.6.post1` or `vllm>=0.8.5` or to create an OpenAI-compatible API endpoint:
|
||||
- SGLang:
|
||||
```shell
|
||||
python -m sglang.launch_server --model-path Qwen/Qwen3-1.7B --reasoning-parser qwen3
|
||||
```
|
||||
- vLLM:
|
||||
```shell
|
||||
vllm serve Qwen/Qwen3-1.7B --enable-reasoning --reasoning-parser deepseek_r1
|
||||
```
|
||||
|
||||
For local use, applications such as Ollama, LMStudio, MLX-LM, llama.cpp, and KTransformers have also supported Qwen3.
|
||||
|
||||
## Switching Between Thinking and Non-Thinking Mode
|
||||
|
||||
> [!TIP]
|
||||
> The `enable_thinking` switch is also available in APIs created by SGLang and vLLM.
|
||||
> Please refer to our documentation for [SGLang](https://qwen.readthedocs.io/en/latest/deployment/sglang.html#thinking-non-thinking-modes) and [vLLM](https://qwen.readthedocs.io/en/latest/deployment/vllm.html#thinking-non-thinking-modes) users.
|
||||
|
||||
### `enable_thinking=True`
|
||||
|
||||
By default, Qwen3 has thinking capabilities enabled, similar to QwQ-32B. This means the model will use its reasoning abilities to enhance the quality of generated responses. For example, when explicitly setting `enable_thinking=True` or leaving it as the default value in `tokenizer.apply_chat_template`, the model will engage its thinking mode.
|
||||
|
||||
```python
|
||||
text = tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
enable_thinking=True # True is the default value for enable_thinking
|
||||
)
|
||||
```
|
||||
|
||||
In this mode, the model will generate think content wrapped in a `<think>...</think>` block, followed by the final response.
|
||||
|
||||
> [!NOTE]
|
||||
> For thinking mode, use `Temperature=0.6`, `TopP=0.95`, `TopK=20`, and `MinP=0` (the default setting in `generation_config.json`). **DO NOT use greedy decoding**, as it can lead to performance degradation and endless repetitions. For more detailed guidance, please refer to the [Best Practices](#best-practices) section.
|
||||
|
||||
|
||||
### `enable_thinking=False`
|
||||
|
||||
We provide a hard switch to strictly disable the model's thinking behavior, aligning its functionality with the previous Qwen2.5-Instruct models. This mode is particularly useful in scenarios where disabling thinking is essential for enhancing efficiency.
|
||||
|
||||
```python
|
||||
text = tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True,
|
||||
enable_thinking=False # Setting enable_thinking=False disables thinking mode
|
||||
)
|
||||
```
|
||||
|
||||
In this mode, the model will not generate any think content and will not include a `<think>...</think>` block.
|
||||
|
||||
> [!NOTE]
|
||||
> For non-thinking mode, we suggest using `Temperature=0.7`, `TopP=0.8`, `TopK=20`, and `MinP=0`. For more detailed guidance, please refer to the [Best Practices](#best-practices) section.
|
||||
|
||||
### Advanced Usage: Switching Between Thinking and Non-Thinking Modes via User Input
|
||||
|
||||
We provide a soft switch mechanism that allows users to dynamically control the model's behavior when `enable_thinking=True`. Specifically, you can add `/think` and `/no_think` to user prompts or system messages to switch the model's thinking mode from turn to turn. The model will follow the most recent instruction in multi-turn conversations.
|
||||
|
||||
Here is an example of a multi-turn conversation:
|
||||
|
||||
```python
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
|
||||
class QwenChatbot:
|
||||
def __init__(self, model_name="Qwen/Qwen3-1.7B"):
|
||||
self.tokenizer = AutoTokenizer.from_pretrained(model_name)
|
||||
self.model = AutoModelForCausalLM.from_pretrained(model_name)
|
||||
self.history = []
|
||||
|
||||
def generate_response(self, user_input):
|
||||
messages = self.history + [{"role": "user", "content": user_input}]
|
||||
|
||||
text = self.tokenizer.apply_chat_template(
|
||||
messages,
|
||||
tokenize=False,
|
||||
add_generation_prompt=True
|
||||
)
|
||||
|
||||
inputs = self.tokenizer(text, return_tensors="pt")
|
||||
response_ids = self.model.generate(**inputs, max_new_tokens=32768)[0][len(inputs.input_ids[0]):].tolist()
|
||||
response = self.tokenizer.decode(response_ids, skip_special_tokens=True)
|
||||
|
||||
# Update history
|
||||
self.history.append({"role": "user", "content": user_input})
|
||||
self.history.append({"role": "assistant", "content": response})
|
||||
|
||||
return response
|
||||
|
||||
# Example Usage
|
||||
if __name__ == "__main__":
|
||||
chatbot = QwenChatbot()
|
||||
|
||||
# First input (without /think or /no_think tags, thinking mode is enabled by default)
|
||||
user_input_1 = "How many r's in strawberries?"
|
||||
print(f"User: {user_input_1}")
|
||||
response_1 = chatbot.generate_response(user_input_1)
|
||||
print(f"Bot: {response_1}")
|
||||
print("----------------------")
|
||||
|
||||
# Second input with /no_think
|
||||
user_input_2 = "Then, how many r's in blueberries? /no_think"
|
||||
print(f"User: {user_input_2}")
|
||||
response_2 = chatbot.generate_response(user_input_2)
|
||||
print(f"Bot: {response_2}")
|
||||
print("----------------------")
|
||||
|
||||
# Third input with /think
|
||||
user_input_3 = "Really? /think"
|
||||
print(f"User: {user_input_3}")
|
||||
response_3 = chatbot.generate_response(user_input_3)
|
||||
print(f"Bot: {response_3}")
|
||||
```
|
||||
|
||||
> [!NOTE]
|
||||
> For API compatibility, when `enable_thinking=True`, regardless of whether the user uses `/think` or `/no_think`, the model will always output a block wrapped in `<think>...</think>`. However, the content inside this block may be empty if thinking is disabled.
|
||||
> When `enable_thinking=False`, the soft switches are not valid. Regardless of any `/think` or `/no_think` tags input by the user, the model will not generate think content and will not include a `<think>...</think>` block.
|
||||
|
||||
## Agentic Use
|
||||
|
||||
Qwen3 excels in tool calling capabilities. We recommend using [Qwen-Agent](https://github.com/QwenLM/Qwen-Agent) to make the best use of agentic ability of Qwen3. Qwen-Agent encapsulates tool-calling templates and tool-calling parsers internally, greatly reducing coding complexity.
|
||||
|
||||
To define the available tools, you can use the MCP configuration file, use the integrated tool of Qwen-Agent, or integrate other tools by yourself.
|
||||
```python
|
||||
from qwen_agent.agents import Assistant
|
||||
|
||||
# Define LLM
|
||||
llm_cfg = {
|
||||
'model': 'Qwen3-1.7B',
|
||||
|
||||
# Use the endpoint provided by Alibaba Model Studio:
|
||||
# 'model_type': 'qwen_dashscope',
|
||||
# 'api_key': os.getenv('DASHSCOPE_API_KEY'),
|
||||
|
||||
# Use a custom endpoint compatible with OpenAI API:
|
||||
'model_server': 'http://localhost:8000/v1', # api_base
|
||||
'api_key': 'EMPTY',
|
||||
|
||||
# Other parameters:
|
||||
# 'generate_cfg': {
|
||||
# # Add: When the response content is `<think>this is the thought</think>this is the answer;
|
||||
# # Do not add: When the response has been separated by reasoning_content and content.
|
||||
# 'thought_in_content': True,
|
||||
# },
|
||||
}
|
||||
|
||||
# Define Tools
|
||||
tools = [
|
||||
{'mcpServers': { # You can specify the MCP configuration file
|
||||
'time': {
|
||||
'command': 'uvx',
|
||||
'args': ['mcp-server-time', '--local-timezone=Asia/Shanghai']
|
||||
},
|
||||
"fetch": {
|
||||
"command": "uvx",
|
||||
"args": ["mcp-server-fetch"]
|
||||
}
|
||||
}
|
||||
},
|
||||
'code_interpreter', # Built-in tools
|
||||
]
|
||||
|
||||
# Define Agent
|
||||
bot = Assistant(llm=llm_cfg, function_list=tools)
|
||||
|
||||
# Streaming generation
|
||||
messages = [{'role': 'user', 'content': 'https://qwenlm.github.io/blog/ Introduce the latest developments of Qwen'}]
|
||||
for responses in bot.run(messages=messages):
|
||||
pass
|
||||
print(responses)
|
||||
```
|
||||
|
||||
## Best Practices
|
||||
|
||||
To achieve optimal performance, we recommend the following settings:
|
||||
|
||||
1. **Sampling Parameters**:
|
||||
- For thinking mode (`enable_thinking=True`), use `Temperature=0.6`, `TopP=0.95`, `TopK=20`, and `MinP=0`. **DO NOT use greedy decoding**, as it can lead to performance degradation and endless repetitions.
|
||||
- For non-thinking mode (`enable_thinking=False`), we suggest using `Temperature=0.7`, `TopP=0.8`, `TopK=20`, and `MinP=0`.
|
||||
- For supported frameworks, you can adjust the `presence_penalty` parameter between 0 and 2 to reduce endless repetitions. However, using a higher value may occasionally result in language mixing and a slight decrease in model performance.
|
||||
|
||||
2. **Adequate Output Length**: We recommend using an output length of 32,768 tokens for most queries. For benchmarking on highly complex problems, such as those found in math and programming competitions, we suggest setting the max output length to 38,912 tokens. This provides the model with sufficient space to generate detailed and comprehensive responses, thereby enhancing its overall performance.
|
||||
|
||||
3. **Standardize Output Format**: We recommend using prompts to standardize model outputs when benchmarking.
|
||||
- **Math Problems**: Include "Please reason step by step, and put your final answer within \boxed{}." in the prompt.
|
||||
- **Multiple-Choice Questions**: Add the following JSON structure to the prompt to standardize responses: "Please show your choice in the `answer` field with only the choice letter, e.g., `"answer": "C"`."
|
||||
|
||||
4. **No Thinking Content in History**: In multi-turn conversations, the historical model output should only include the final output part and does not need to include the thinking content. It is implemented in the provided chat template in Jinja2. However, for frameworks that do not directly use the Jinja2 chat template, it is up to the developers to ensure that the best practice is followed.
|
||||
|
||||
### Citation
|
||||
|
||||
If you find our work helpful, feel free to give us a cite.
|
||||
|
||||
```
|
||||
@misc{qwen3technicalreport,
|
||||
title={Qwen3 Technical Report},
|
||||
author={Qwen Team},
|
||||
year={2025},
|
||||
eprint={2505.09388},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CL},
|
||||
url={https://arxiv.org/abs/2505.09388},
|
||||
}
|
||||
```
|
||||
30
config.json
Normal file
30
config.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 151643,
|
||||
"eos_token_id": 151645,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"model_type": "qwen3",
|
||||
"num_attention_heads": 16,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.51.0",
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
context_encoding_model/_tp0_bk0/command.txt
Normal file
1
context_encoding_model/_tp0_bk0/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_bdcbea8455ebae357b4c+541d7181.hlo_module.pb --output model.MODULE_bdcbea8455ebae357b4c+541d7181.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk0/log-neuron-cc.txt"]
|
||||
1147
context_encoding_model/_tp0_bk0/global_metric_store.json
Normal file
1147
context_encoding_model/_tp0_bk0/global_metric_store.json
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk0/graph.neff
Normal file
3
context_encoding_model/_tp0_bk0/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:92d81f134ce62a3b1bcb9c577d19d5ea4c462cc6a79b0b0e499d1560e903979b
|
||||
size 830464
|
||||
9211
context_encoding_model/_tp0_bk0/log-neuron-cc.txt
Normal file
9211
context_encoding_model/_tp0_bk0/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk0/metaneff.pb
Normal file
3
context_encoding_model/_tp0_bk0/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ec4b0c42828e50013dc0bd682bdb81c0aa36920c9456fd53a653c18581ad49d1
|
||||
size 1877151
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5239d1061207e7bd19a6be27bb12fddcd4b60b66cb5ba20fb4ab47a4be30d4c5
|
||||
size 1961858
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:92d81f134ce62a3b1bcb9c577d19d5ea4c462cc6a79b0b0e499d1560e903979b
|
||||
size 830464
|
||||
224
context_encoding_model/_tp0_bk0/neuron_config.json
Normal file
224
context_encoding_model/_tp0_bk0/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 1,
|
||||
"bucket_n_active_tokens": true,
|
||||
"buckets": [
|
||||
128
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 2,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": [
|
||||
128
|
||||
],
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": true,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 2048,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": null,
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
context_encoding_model/_tp0_bk1/command.txt
Normal file
1
context_encoding_model/_tp0_bk1/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_e7dad336ed1c266a3016+bbc3fa47.hlo_module.pb --output model.MODULE_e7dad336ed1c266a3016+bbc3fa47.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk1/log-neuron-cc.txt"]
|
||||
1177
context_encoding_model/_tp0_bk1/global_metric_store.json
Normal file
1177
context_encoding_model/_tp0_bk1/global_metric_store.json
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk1/graph.neff
Normal file
3
context_encoding_model/_tp0_bk1/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ca594c9d5f303adb5de3e5aa58b2eca9ac1ed3aff762e1b09620d6474407d8bc
|
||||
size 861184
|
||||
9530
context_encoding_model/_tp0_bk1/log-neuron-cc.txt
Normal file
9530
context_encoding_model/_tp0_bk1/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk1/metaneff.pb
Normal file
3
context_encoding_model/_tp0_bk1/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:20425c1f386fac6326c8423f6243a37818ce0b36840edfe0fdf47dc9a74af5f2
|
||||
size 2194530
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4dbd208cf8cb5cfeef58be408e28cdbf0a23d70a7cd3affd1510ceae9fb95c8b
|
||||
size 2280924
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ca594c9d5f303adb5de3e5aa58b2eca9ac1ed3aff762e1b09620d6474407d8bc
|
||||
size 861184
|
||||
224
context_encoding_model/_tp0_bk1/neuron_config.json
Normal file
224
context_encoding_model/_tp0_bk1/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 1,
|
||||
"bucket_n_active_tokens": true,
|
||||
"buckets": [
|
||||
256
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 2,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": [
|
||||
256
|
||||
],
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": true,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 2048,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": null,
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
context_encoding_model/_tp0_bk2/command.txt
Normal file
1
context_encoding_model/_tp0_bk2/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_f36c9ad51e28c98c9723+c4081b94.hlo_module.pb --output model.MODULE_f36c9ad51e28c98c9723+c4081b94.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk2/log-neuron-cc.txt"]
|
||||
1177
context_encoding_model/_tp0_bk2/global_metric_store.json
Normal file
1177
context_encoding_model/_tp0_bk2/global_metric_store.json
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk2/graph.neff
Normal file
3
context_encoding_model/_tp0_bk2/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f7292240129556fa9d8d8412b12947fe2577cd0c6367f047d23e7e4526896cce
|
||||
size 912384
|
||||
9521
context_encoding_model/_tp0_bk2/log-neuron-cc.txt
Normal file
9521
context_encoding_model/_tp0_bk2/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk2/metaneff.pb
Normal file
3
context_encoding_model/_tp0_bk2/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3a52b69e5f1f8c87121e293c72dfb7fb9dc3f8050a04452be5e7691addcc3371
|
||||
size 2280546
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:89ed0e9df247f2125a989d4e0bbc25ce61ee24d187117a143bd2bd583a741239
|
||||
size 2366940
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f7292240129556fa9d8d8412b12947fe2577cd0c6367f047d23e7e4526896cce
|
||||
size 912384
|
||||
224
context_encoding_model/_tp0_bk2/neuron_config.json
Normal file
224
context_encoding_model/_tp0_bk2/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 1,
|
||||
"bucket_n_active_tokens": true,
|
||||
"buckets": [
|
||||
512
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 2,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": [
|
||||
512
|
||||
],
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": true,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 2048,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": null,
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
context_encoding_model/_tp0_bk3/command.txt
Normal file
1
context_encoding_model/_tp0_bk3/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_0984f4c19a044cc11c2a+1c024d2c.hlo_module.pb --output model.MODULE_0984f4c19a044cc11c2a+1c024d2c.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk3/log-neuron-cc.txt"]
|
||||
1177
context_encoding_model/_tp0_bk3/global_metric_store.json
Normal file
1177
context_encoding_model/_tp0_bk3/global_metric_store.json
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk3/graph.neff
Normal file
3
context_encoding_model/_tp0_bk3/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:88e20e9aacad85427bc64b564935a053a2d089c921c32c96c0f2363c14b1f8fb
|
||||
size 1004544
|
||||
9514
context_encoding_model/_tp0_bk3/log-neuron-cc.txt
Normal file
9514
context_encoding_model/_tp0_bk3/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk3/metaneff.pb
Normal file
3
context_encoding_model/_tp0_bk3/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f4bfe18bacca09ec836a16458059ad676212bf8842fbb89645dbe0b21f90bea0
|
||||
size 2454034
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:64bf60998b259cee8443492d313f8dd44a9b30e8f7465266150157a423e9164c
|
||||
size 2540428
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:88e20e9aacad85427bc64b564935a053a2d089c921c32c96c0f2363c14b1f8fb
|
||||
size 1004544
|
||||
224
context_encoding_model/_tp0_bk3/neuron_config.json
Normal file
224
context_encoding_model/_tp0_bk3/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 1,
|
||||
"bucket_n_active_tokens": true,
|
||||
"buckets": [
|
||||
1024
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 2,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": [
|
||||
1024
|
||||
],
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": true,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 2048,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": null,
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
context_encoding_model/_tp0_bk4/command.txt
Normal file
1
context_encoding_model/_tp0_bk4/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_e6b44ff520e6c4333666+e9aa1481.hlo_module.pb --output model.MODULE_e6b44ff520e6c4333666+e9aa1481.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk4/log-neuron-cc.txt"]
|
||||
1177
context_encoding_model/_tp0_bk4/global_metric_store.json
Normal file
1177
context_encoding_model/_tp0_bk4/global_metric_store.json
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk4/graph.neff
Normal file
3
context_encoding_model/_tp0_bk4/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f1666f119851fd3de7f6ea4cb5da22169bcf4b1284acdd2f9f3afdc8e33da8b0
|
||||
size 1250304
|
||||
9501
context_encoding_model/_tp0_bk4/log-neuron-cc.txt
Normal file
9501
context_encoding_model/_tp0_bk4/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
context_encoding_model/_tp0_bk4/metaneff.pb
Normal file
3
context_encoding_model/_tp0_bk4/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:352da1073d15c0fbd9355ed2074a8ee0012540534b027fe615c424d66b1ee859
|
||||
size 2798098
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:91f063d37769930b9be102cb6a84963b790e2ac12eb75d4e6018c2b479cf533d
|
||||
size 2884492
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f1666f119851fd3de7f6ea4cb5da22169bcf4b1284acdd2f9f3afdc8e33da8b0
|
||||
size 1250304
|
||||
224
context_encoding_model/_tp0_bk4/neuron_config.json
Normal file
224
context_encoding_model/_tp0_bk4/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 1,
|
||||
"bucket_n_active_tokens": true,
|
||||
"buckets": [
|
||||
2048
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 2,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": [
|
||||
2048
|
||||
],
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": true,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 2048,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": null,
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
13
generation_config.json
Normal file
13
generation_config.json
Normal file
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"bos_token_id": 151643,
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151645,
|
||||
151643
|
||||
],
|
||||
"pad_token_id": 151643,
|
||||
"temperature": 0.6,
|
||||
"top_k": 20,
|
||||
"top_p": 0.95,
|
||||
"transformers_version": "4.51.0"
|
||||
}
|
||||
1
layout_opt/command.txt
Normal file
1
layout_opt/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile graph.hlo --framework XLA --target trn2 --output graph.neff --model-type=transformer -O1 --lnc=2 '--internal-hlo2tensorizer-options=--experimental-unsafe-fp8e4m3fn-as-fp8e4m3 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35
|
||||
3
layout_opt/graph.neff
Normal file
3
layout_opt/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8c8d68c65608dd3d8d871b6447ca2ef8a60df383cadfef8cbefdcc0391f53c10
|
||||
size 1055744
|
||||
3308
layout_opt/log-neuron-cc.txt
Normal file
3308
layout_opt/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
934
layout_opt/metaneff
Normal file
934
layout_opt/metaneff
Normal file
@@ -0,0 +1,934 @@
|
||||
|
||||
(
|
||||
input0<05><> <09>2embed_tokens.weight8
|
||||
;
|
||||
input1<04><10>2'layers.0.self_attn.o_proj.o_proj.weight8
|
||||
=
|
||||
input2<04><02>2)layers.0.self_attn.qkv_proj.v_proj.weight8
|
||||
1
|
||||
input3<02>2layers.0.input_layernorm.weight8
|
||||
7
|
||||
input4<02>2%layers.0.self_attn.k_layernorm.weight8
|
||||
=
|
||||
input5<04><02>2)layers.0.self_attn.qkv_proj.k_proj.weight8
|
||||
7
|
||||
input6<02>2%layers.0.self_attn.q_layernorm.weight8
|
||||
=
|
||||
input7<04><04>2)layers.0.self_attn.qkv_proj.q_proj.weight8
|
||||
1
|
||||
input8<04><10>2layers.0.mlp.down_proj.weight8
|
||||
/
|
||||
input9<04><0C>2layers.0.mlp.up_proj.weight8
|
||||
;
|
||||
input10<02>2(layers.0.post_attention_layernorm.weight8
|
||||
2
|
||||
input11<04><0C>2layers.0.mlp.gate_proj.weight8
|
||||
<
|
||||
input12<04><10>2'layers.1.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input13<04><02>2)layers.1.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input14<02>2layers.1.input_layernorm.weight8
|
||||
8
|
||||
input15<02>2%layers.1.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input16<04><02>2)layers.1.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input17<02>2%layers.1.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input18<04><04>2)layers.1.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input19<04><10>2layers.1.mlp.down_proj.weight8
|
||||
0
|
||||
input20<04><0C>2layers.1.mlp.up_proj.weight8
|
||||
;
|
||||
input21<02>2(layers.1.post_attention_layernorm.weight8
|
||||
2
|
||||
input22<04><0C>2layers.1.mlp.gate_proj.weight8
|
||||
<
|
||||
input23<04><10>2'layers.2.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input24<04><02>2)layers.2.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input25<02>2layers.2.input_layernorm.weight8
|
||||
8
|
||||
input26<02>2%layers.2.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input27<04><02>2)layers.2.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input28<02>2%layers.2.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input29<04><04>2)layers.2.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input30<04><10>2layers.2.mlp.down_proj.weight8
|
||||
0
|
||||
input31<04><0C>2layers.2.mlp.up_proj.weight8
|
||||
;
|
||||
input32<02>2(layers.2.post_attention_layernorm.weight8
|
||||
2
|
||||
input33<04><0C>2layers.2.mlp.gate_proj.weight8
|
||||
<
|
||||
input34<04><10>2'layers.3.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input35<04><02>2)layers.3.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input36<02>2layers.3.input_layernorm.weight8
|
||||
8
|
||||
input37<02>2%layers.3.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input38<04><02>2)layers.3.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input39<02>2%layers.3.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input40<04><04>2)layers.3.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input41<04><10>2layers.3.mlp.down_proj.weight8
|
||||
0
|
||||
input42<04><0C>2layers.3.mlp.up_proj.weight8
|
||||
;
|
||||
input43<02>2(layers.3.post_attention_layernorm.weight8
|
||||
2
|
||||
input44<04><0C>2layers.3.mlp.gate_proj.weight8
|
||||
<
|
||||
input45<04><10>2'layers.4.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input46<04><02>2)layers.4.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input47<02>2layers.4.input_layernorm.weight8
|
||||
8
|
||||
input48<02>2%layers.4.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input49<04><02>2)layers.4.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input50<02>2%layers.4.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input51<04><04>2)layers.4.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input52<04><10>2layers.4.mlp.down_proj.weight8
|
||||
0
|
||||
input53<04><0C>2layers.4.mlp.up_proj.weight8
|
||||
;
|
||||
input54<02>2(layers.4.post_attention_layernorm.weight8
|
||||
2
|
||||
input55<04><0C>2layers.4.mlp.gate_proj.weight8
|
||||
<
|
||||
input56<04><10>2'layers.5.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input57<04><02>2)layers.5.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input58<02>2layers.5.input_layernorm.weight8
|
||||
8
|
||||
input59<02>2%layers.5.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input60<04><02>2)layers.5.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input61<02>2%layers.5.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input62<04><04>2)layers.5.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input63<04><10>2layers.5.mlp.down_proj.weight8
|
||||
0
|
||||
input64<04><0C>2layers.5.mlp.up_proj.weight8
|
||||
;
|
||||
input65<02>2(layers.5.post_attention_layernorm.weight8
|
||||
2
|
||||
input66<04><0C>2layers.5.mlp.gate_proj.weight8
|
||||
<
|
||||
input67<04><10>2'layers.6.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input68<04><02>2)layers.6.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input69<02>2layers.6.input_layernorm.weight8
|
||||
8
|
||||
input70<02>2%layers.6.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input71<04><02>2)layers.6.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input72<02>2%layers.6.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input73<04><04>2)layers.6.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input74<04><10>2layers.6.mlp.down_proj.weight8
|
||||
0
|
||||
input75<04><0C>2layers.6.mlp.up_proj.weight8
|
||||
;
|
||||
input76<02>2(layers.6.post_attention_layernorm.weight8
|
||||
2
|
||||
input77<04><0C>2layers.6.mlp.gate_proj.weight8
|
||||
<
|
||||
input78<04><10>2'layers.7.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input79<04><02>2)layers.7.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input80<02>2layers.7.input_layernorm.weight8
|
||||
8
|
||||
input81<02>2%layers.7.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input82<04><02>2)layers.7.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input83<02>2%layers.7.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input84<04><04>2)layers.7.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input85<04><10>2layers.7.mlp.down_proj.weight8
|
||||
0
|
||||
input86<04><0C>2layers.7.mlp.up_proj.weight8
|
||||
;
|
||||
input87<02>2(layers.7.post_attention_layernorm.weight8
|
||||
2
|
||||
input88<04><0C>2layers.7.mlp.gate_proj.weight8
|
||||
<
|
||||
input89<04><10>2'layers.8.self_attn.o_proj.o_proj.weight8
|
||||
>
|
||||
input90<04><02>2)layers.8.self_attn.qkv_proj.v_proj.weight8
|
||||
2
|
||||
input91<02>2layers.8.input_layernorm.weight8
|
||||
8
|
||||
input92<02>2%layers.8.self_attn.k_layernorm.weight8
|
||||
>
|
||||
input93<04><02>2)layers.8.self_attn.qkv_proj.k_proj.weight8
|
||||
8
|
||||
input94<02>2%layers.8.self_attn.q_layernorm.weight8
|
||||
>
|
||||
input95<04><04>2)layers.8.self_attn.qkv_proj.q_proj.weight8
|
||||
2
|
||||
input96<04><10>2layers.8.mlp.down_proj.weight8
|
||||
0
|
||||
input97<04><0C>2layers.8.mlp.up_proj.weight8
|
||||
;
|
||||
input98<02>2(layers.8.post_attention_layernorm.weight8
|
||||
2
|
||||
input99<04><0C>2layers.8.mlp.gate_proj.weight8
|
||||
=
|
||||
input100<04><10>2'layers.9.self_attn.o_proj.o_proj.weight8
|
||||
?
|
||||
input101<04><02>2)layers.9.self_attn.qkv_proj.v_proj.weight8
|
||||
3
|
||||
input102<02>2layers.9.input_layernorm.weight8
|
||||
9
|
||||
input103<02>2%layers.9.self_attn.k_layernorm.weight8
|
||||
?
|
||||
input104<04><02>2)layers.9.self_attn.qkv_proj.k_proj.weight8
|
||||
9
|
||||
input105<02>2%layers.9.self_attn.q_layernorm.weight8
|
||||
?
|
||||
input106<04><04>2)layers.9.self_attn.qkv_proj.q_proj.weight8
|
||||
3
|
||||
input107<04><10>2layers.9.mlp.down_proj.weight8
|
||||
1
|
||||
input108<04><0C>2layers.9.mlp.up_proj.weight8
|
||||
<
|
||||
input109<02>2(layers.9.post_attention_layernorm.weight8
|
||||
3
|
||||
input110<04><0C>2layers.9.mlp.gate_proj.weight8
|
||||
>
|
||||
input111<04><10>2(layers.10.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input112<04><02>2*layers.10.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input113<02>2 layers.10.input_layernorm.weight8
|
||||
:
|
||||
input114<02>2&layers.10.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input115<04><02>2*layers.10.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input116<02>2&layers.10.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input117<04><04>2*layers.10.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input118<04><10>2layers.10.mlp.down_proj.weight8
|
||||
2
|
||||
input119<04><0C>2layers.10.mlp.up_proj.weight8
|
||||
=
|
||||
input120<02>2)layers.10.post_attention_layernorm.weight8
|
||||
4
|
||||
input121<04><0C>2layers.10.mlp.gate_proj.weight8
|
||||
>
|
||||
input122<04><10>2(layers.11.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input123<04><02>2*layers.11.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input124<02>2 layers.11.input_layernorm.weight8
|
||||
:
|
||||
input125<02>2&layers.11.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input126<04><02>2*layers.11.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input127<02>2&layers.11.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input128<04><04>2*layers.11.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input129<04><10>2layers.11.mlp.down_proj.weight8
|
||||
2
|
||||
input130<04><0C>2layers.11.mlp.up_proj.weight8
|
||||
=
|
||||
input131<02>2)layers.11.post_attention_layernorm.weight8
|
||||
4
|
||||
input132<04><0C>2layers.11.mlp.gate_proj.weight8
|
||||
>
|
||||
input133<04><10>2(layers.12.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input134<04><02>2*layers.12.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input135<02>2 layers.12.input_layernorm.weight8
|
||||
:
|
||||
input136<02>2&layers.12.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input137<04><02>2*layers.12.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input138<02>2&layers.12.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input139<04><04>2*layers.12.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input140<04><10>2layers.12.mlp.down_proj.weight8
|
||||
2
|
||||
input141<04><0C>2layers.12.mlp.up_proj.weight8
|
||||
=
|
||||
input142<02>2)layers.12.post_attention_layernorm.weight8
|
||||
4
|
||||
input143<04><0C>2layers.12.mlp.gate_proj.weight8
|
||||
>
|
||||
input144<04><10>2(layers.13.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input145<04><02>2*layers.13.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input146<02>2 layers.13.input_layernorm.weight8
|
||||
:
|
||||
input147<02>2&layers.13.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input148<04><02>2*layers.13.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input149<02>2&layers.13.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input150<04><04>2*layers.13.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input151<04><10>2layers.13.mlp.down_proj.weight8
|
||||
2
|
||||
input152<04><0C>2layers.13.mlp.up_proj.weight8
|
||||
=
|
||||
input153<02>2)layers.13.post_attention_layernorm.weight8
|
||||
4
|
||||
input154<04><0C>2layers.13.mlp.gate_proj.weight8
|
||||
>
|
||||
input155<04><10>2(layers.14.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input156<04><02>2*layers.14.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input157<02>2 layers.14.input_layernorm.weight8
|
||||
:
|
||||
input158<02>2&layers.14.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input159<04><02>2*layers.14.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input160<02>2&layers.14.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input161<04><04>2*layers.14.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input162<04><10>2layers.14.mlp.down_proj.weight8
|
||||
2
|
||||
input163<04><0C>2layers.14.mlp.up_proj.weight8
|
||||
=
|
||||
input164<02>2)layers.14.post_attention_layernorm.weight8
|
||||
4
|
||||
input165<04><0C>2layers.14.mlp.gate_proj.weight8
|
||||
>
|
||||
input166<04><10>2(layers.15.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input167<04><02>2*layers.15.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input168<02>2 layers.15.input_layernorm.weight8
|
||||
:
|
||||
input169<02>2&layers.15.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input170<04><02>2*layers.15.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input171<02>2&layers.15.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input172<04><04>2*layers.15.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input173<04><10>2layers.15.mlp.down_proj.weight8
|
||||
2
|
||||
input174<04><0C>2layers.15.mlp.up_proj.weight8
|
||||
=
|
||||
input175<02>2)layers.15.post_attention_layernorm.weight8
|
||||
4
|
||||
input176<04><0C>2layers.15.mlp.gate_proj.weight8
|
||||
>
|
||||
input177<04><10>2(layers.16.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input178<04><02>2*layers.16.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input179<02>2 layers.16.input_layernorm.weight8
|
||||
:
|
||||
input180<02>2&layers.16.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input181<04><02>2*layers.16.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input182<02>2&layers.16.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input183<04><04>2*layers.16.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input184<04><10>2layers.16.mlp.down_proj.weight8
|
||||
2
|
||||
input185<04><0C>2layers.16.mlp.up_proj.weight8
|
||||
=
|
||||
input186<02>2)layers.16.post_attention_layernorm.weight8
|
||||
4
|
||||
input187<04><0C>2layers.16.mlp.gate_proj.weight8
|
||||
>
|
||||
input188<04><10>2(layers.17.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input189<04><02>2*layers.17.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input190<02>2 layers.17.input_layernorm.weight8
|
||||
:
|
||||
input191<02>2&layers.17.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input192<04><02>2*layers.17.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input193<02>2&layers.17.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input194<04><04>2*layers.17.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input195<04><10>2layers.17.mlp.down_proj.weight8
|
||||
2
|
||||
input196<04><0C>2layers.17.mlp.up_proj.weight8
|
||||
=
|
||||
input197<02>2)layers.17.post_attention_layernorm.weight8
|
||||
4
|
||||
input198<04><0C>2layers.17.mlp.gate_proj.weight8
|
||||
>
|
||||
input199<04><10>2(layers.18.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input200<04><02>2*layers.18.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input201<02>2 layers.18.input_layernorm.weight8
|
||||
:
|
||||
input202<02>2&layers.18.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input203<04><02>2*layers.18.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input204<02>2&layers.18.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input205<04><04>2*layers.18.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input206<04><10>2layers.18.mlp.down_proj.weight8
|
||||
2
|
||||
input207<04><0C>2layers.18.mlp.up_proj.weight8
|
||||
=
|
||||
input208<02>2)layers.18.post_attention_layernorm.weight8
|
||||
4
|
||||
input209<04><0C>2layers.18.mlp.gate_proj.weight8
|
||||
>
|
||||
input210<04><10>2(layers.19.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input211<04><02>2*layers.19.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input212<02>2 layers.19.input_layernorm.weight8
|
||||
:
|
||||
input213<02>2&layers.19.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input214<04><02>2*layers.19.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input215<02>2&layers.19.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input216<04><04>2*layers.19.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input217<04><10>2layers.19.mlp.down_proj.weight8
|
||||
2
|
||||
input218<04><0C>2layers.19.mlp.up_proj.weight8
|
||||
=
|
||||
input219<02>2)layers.19.post_attention_layernorm.weight8
|
||||
4
|
||||
input220<04><0C>2layers.19.mlp.gate_proj.weight8
|
||||
>
|
||||
input221<04><10>2(layers.20.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input222<04><02>2*layers.20.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input223<02>2 layers.20.input_layernorm.weight8
|
||||
:
|
||||
input224<02>2&layers.20.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input225<04><02>2*layers.20.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input226<02>2&layers.20.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input227<04><04>2*layers.20.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input228<04><10>2layers.20.mlp.down_proj.weight8
|
||||
2
|
||||
input229<04><0C>2layers.20.mlp.up_proj.weight8
|
||||
=
|
||||
input230<02>2)layers.20.post_attention_layernorm.weight8
|
||||
4
|
||||
input231<04><0C>2layers.20.mlp.gate_proj.weight8
|
||||
>
|
||||
input232<04><10>2(layers.21.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input233<04><02>2*layers.21.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input234<02>2 layers.21.input_layernorm.weight8
|
||||
:
|
||||
input235<02>2&layers.21.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input236<04><02>2*layers.21.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input237<02>2&layers.21.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input238<04><04>2*layers.21.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input239<04><10>2layers.21.mlp.down_proj.weight8
|
||||
2
|
||||
input240<04><0C>2layers.21.mlp.up_proj.weight8
|
||||
=
|
||||
input241<02>2)layers.21.post_attention_layernorm.weight8
|
||||
4
|
||||
input242<04><0C>2layers.21.mlp.gate_proj.weight8
|
||||
>
|
||||
input243<04><10>2(layers.22.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input244<04><02>2*layers.22.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input245<02>2 layers.22.input_layernorm.weight8
|
||||
:
|
||||
input246<02>2&layers.22.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input247<04><02>2*layers.22.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input248<02>2&layers.22.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input249<04><04>2*layers.22.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input250<04><10>2layers.22.mlp.down_proj.weight8
|
||||
2
|
||||
input251<04><0C>2layers.22.mlp.up_proj.weight8
|
||||
=
|
||||
input252<02>2)layers.22.post_attention_layernorm.weight8
|
||||
4
|
||||
input253<04><0C>2layers.22.mlp.gate_proj.weight8
|
||||
>
|
||||
input254<04><10>2(layers.23.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input255<04><02>2*layers.23.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input256<02>2 layers.23.input_layernorm.weight8
|
||||
:
|
||||
input257<02>2&layers.23.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input258<04><02>2*layers.23.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input259<02>2&layers.23.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input260<04><04>2*layers.23.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input261<04><10>2layers.23.mlp.down_proj.weight8
|
||||
2
|
||||
input262<04><0C>2layers.23.mlp.up_proj.weight8
|
||||
=
|
||||
input263<02>2)layers.23.post_attention_layernorm.weight8
|
||||
4
|
||||
input264<04><0C>2layers.23.mlp.gate_proj.weight8
|
||||
>
|
||||
input265<04><10>2(layers.24.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input266<04><02>2*layers.24.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input267<02>2 layers.24.input_layernorm.weight8
|
||||
:
|
||||
input268<02>2&layers.24.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input269<04><02>2*layers.24.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input270<02>2&layers.24.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input271<04><04>2*layers.24.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input272<04><10>2layers.24.mlp.down_proj.weight8
|
||||
2
|
||||
input273<04><0C>2layers.24.mlp.up_proj.weight8
|
||||
=
|
||||
input274<02>2)layers.24.post_attention_layernorm.weight8
|
||||
4
|
||||
input275<04><0C>2layers.24.mlp.gate_proj.weight8
|
||||
>
|
||||
input276<04><10>2(layers.25.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input277<04><02>2*layers.25.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input278<02>2 layers.25.input_layernorm.weight8
|
||||
:
|
||||
input279<02>2&layers.25.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input280<04><02>2*layers.25.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input281<02>2&layers.25.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input282<04><04>2*layers.25.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input283<04><10>2layers.25.mlp.down_proj.weight8
|
||||
2
|
||||
input284<04><0C>2layers.25.mlp.up_proj.weight8
|
||||
=
|
||||
input285<02>2)layers.25.post_attention_layernorm.weight8
|
||||
4
|
||||
input286<04><0C>2layers.25.mlp.gate_proj.weight8
|
||||
>
|
||||
input287<04><10>2(layers.26.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input288<04><02>2*layers.26.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input289<02>2 layers.26.input_layernorm.weight8
|
||||
:
|
||||
input290<02>2&layers.26.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input291<04><02>2*layers.26.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input292<02>2&layers.26.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input293<04><04>2*layers.26.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input294<04><10>2layers.26.mlp.down_proj.weight8
|
||||
2
|
||||
input295<04><0C>2layers.26.mlp.up_proj.weight8
|
||||
=
|
||||
input296<02>2)layers.26.post_attention_layernorm.weight8
|
||||
4
|
||||
input297<04><0C>2layers.26.mlp.gate_proj.weight8
|
||||
>
|
||||
input298<04><10>2(layers.27.self_attn.o_proj.o_proj.weight8
|
||||
@
|
||||
input299<04><02>2*layers.27.self_attn.qkv_proj.v_proj.weight8
|
||||
4
|
||||
input300<02>2 layers.27.input_layernorm.weight8
|
||||
:
|
||||
input301<02>2&layers.27.self_attn.k_layernorm.weight8
|
||||
@
|
||||
input302<04><02>2*layers.27.self_attn.qkv_proj.k_proj.weight8
|
||||
:
|
||||
input303<02>2&layers.27.self_attn.q_layernorm.weight8
|
||||
@
|
||||
input304<04><04>2*layers.27.self_attn.qkv_proj.q_proj.weight8
|
||||
4
|
||||
input305<04><10>2layers.27.mlp.down_proj.weight8
|
||||
2
|
||||
input306<04><0C>2layers.27.mlp.up_proj.weight8
|
||||
=
|
||||
input307<02>2)layers.27.post_attention_layernorm.weight8
|
||||
4
|
||||
input308<04><0C>2layers.27.mlp.gate_proj.weight8
|
||||
%
|
||||
input309<05><><02>2lm_head.weight8
|
||||
|
||||
input310<02>2norm.weight8'
|
||||
output0<05><> <09>2embed_tokens.weight:
|
||||
output1<04><10>2'layers.0.self_attn.o_proj.o_proj.weight<
|
||||
output2<04><02>2)layers.0.self_attn.qkv_proj.v_proj.weight0
|
||||
output3<02>2layers.0.input_layernorm.weight6
|
||||
output4<02>2%layers.0.self_attn.k_layernorm.weight<
|
||||
output5<04><02>2)layers.0.self_attn.qkv_proj.k_proj.weight6
|
||||
output6<02>2%layers.0.self_attn.q_layernorm.weight<
|
||||
output7<04><04>2)layers.0.self_attn.qkv_proj.q_proj.weight0
|
||||
output8<04><10>2layers.0.mlp.down_proj.weight.
|
||||
output9<04><0C>2layers.0.mlp.up_proj.weight:
|
||||
output10<02>2(layers.0.post_attention_layernorm.weight1
|
||||
output11<04><0C>2layers.0.mlp.gate_proj.weight;
|
||||
output12<04><10>2'layers.1.self_attn.o_proj.o_proj.weight=
|
||||
output13<04><02>2)layers.1.self_attn.qkv_proj.v_proj.weight1
|
||||
output14<02>2layers.1.input_layernorm.weight7
|
||||
output15<02>2%layers.1.self_attn.k_layernorm.weight=
|
||||
output16<04><02>2)layers.1.self_attn.qkv_proj.k_proj.weight7
|
||||
output17<02>2%layers.1.self_attn.q_layernorm.weight=
|
||||
output18<04><04>2)layers.1.self_attn.qkv_proj.q_proj.weight1
|
||||
output19<04><10>2layers.1.mlp.down_proj.weight/
|
||||
output20<04><0C>2layers.1.mlp.up_proj.weight:
|
||||
output21<02>2(layers.1.post_attention_layernorm.weight1
|
||||
output22<04><0C>2layers.1.mlp.gate_proj.weight;
|
||||
output23<04><10>2'layers.2.self_attn.o_proj.o_proj.weight=
|
||||
output24<04><02>2)layers.2.self_attn.qkv_proj.v_proj.weight1
|
||||
output25<02>2layers.2.input_layernorm.weight7
|
||||
output26<02>2%layers.2.self_attn.k_layernorm.weight=
|
||||
output27<04><02>2)layers.2.self_attn.qkv_proj.k_proj.weight7
|
||||
output28<02>2%layers.2.self_attn.q_layernorm.weight=
|
||||
output29<04><04>2)layers.2.self_attn.qkv_proj.q_proj.weight1
|
||||
output30<04><10>2layers.2.mlp.down_proj.weight/
|
||||
output31<04><0C>2layers.2.mlp.up_proj.weight:
|
||||
output32<02>2(layers.2.post_attention_layernorm.weight1
|
||||
output33<04><0C>2layers.2.mlp.gate_proj.weight;
|
||||
output34<04><10>2'layers.3.self_attn.o_proj.o_proj.weight=
|
||||
output35<04><02>2)layers.3.self_attn.qkv_proj.v_proj.weight1
|
||||
output36<02>2layers.3.input_layernorm.weight7
|
||||
output37<02>2%layers.3.self_attn.k_layernorm.weight=
|
||||
output38<04><02>2)layers.3.self_attn.qkv_proj.k_proj.weight7
|
||||
output39<02>2%layers.3.self_attn.q_layernorm.weight=
|
||||
output40<04><04>2)layers.3.self_attn.qkv_proj.q_proj.weight1
|
||||
output41<04><10>2layers.3.mlp.down_proj.weight/
|
||||
output42<04><0C>2layers.3.mlp.up_proj.weight:
|
||||
output43<02>2(layers.3.post_attention_layernorm.weight1
|
||||
output44<04><0C>2layers.3.mlp.gate_proj.weight;
|
||||
output45<04><10>2'layers.4.self_attn.o_proj.o_proj.weight=
|
||||
output46<04><02>2)layers.4.self_attn.qkv_proj.v_proj.weight1
|
||||
output47<02>2layers.4.input_layernorm.weight7
|
||||
output48<02>2%layers.4.self_attn.k_layernorm.weight=
|
||||
output49<04><02>2)layers.4.self_attn.qkv_proj.k_proj.weight7
|
||||
output50<02>2%layers.4.self_attn.q_layernorm.weight=
|
||||
output51<04><04>2)layers.4.self_attn.qkv_proj.q_proj.weight1
|
||||
output52<04><10>2layers.4.mlp.down_proj.weight/
|
||||
output53<04><0C>2layers.4.mlp.up_proj.weight:
|
||||
output54<02>2(layers.4.post_attention_layernorm.weight1
|
||||
output55<04><0C>2layers.4.mlp.gate_proj.weight;
|
||||
output56<04><10>2'layers.5.self_attn.o_proj.o_proj.weight=
|
||||
output57<04><02>2)layers.5.self_attn.qkv_proj.v_proj.weight1
|
||||
output58<02>2layers.5.input_layernorm.weight7
|
||||
output59<02>2%layers.5.self_attn.k_layernorm.weight=
|
||||
output60<04><02>2)layers.5.self_attn.qkv_proj.k_proj.weight7
|
||||
output61<02>2%layers.5.self_attn.q_layernorm.weight=
|
||||
output62<04><04>2)layers.5.self_attn.qkv_proj.q_proj.weight1
|
||||
output63<04><10>2layers.5.mlp.down_proj.weight/
|
||||
output64<04><0C>2layers.5.mlp.up_proj.weight:
|
||||
output65<02>2(layers.5.post_attention_layernorm.weight1
|
||||
output66<04><0C>2layers.5.mlp.gate_proj.weight;
|
||||
output67<04><10>2'layers.6.self_attn.o_proj.o_proj.weight=
|
||||
output68<04><02>2)layers.6.self_attn.qkv_proj.v_proj.weight1
|
||||
output69<02>2layers.6.input_layernorm.weight7
|
||||
output70<02>2%layers.6.self_attn.k_layernorm.weight=
|
||||
output71<04><02>2)layers.6.self_attn.qkv_proj.k_proj.weight7
|
||||
output72<02>2%layers.6.self_attn.q_layernorm.weight=
|
||||
output73<04><04>2)layers.6.self_attn.qkv_proj.q_proj.weight1
|
||||
output74<04><10>2layers.6.mlp.down_proj.weight/
|
||||
output75<04><0C>2layers.6.mlp.up_proj.weight:
|
||||
output76<02>2(layers.6.post_attention_layernorm.weight1
|
||||
output77<04><0C>2layers.6.mlp.gate_proj.weight;
|
||||
output78<04><10>2'layers.7.self_attn.o_proj.o_proj.weight=
|
||||
output79<04><02>2)layers.7.self_attn.qkv_proj.v_proj.weight1
|
||||
output80<02>2layers.7.input_layernorm.weight7
|
||||
output81<02>2%layers.7.self_attn.k_layernorm.weight=
|
||||
output82<04><02>2)layers.7.self_attn.qkv_proj.k_proj.weight7
|
||||
output83<02>2%layers.7.self_attn.q_layernorm.weight=
|
||||
output84<04><04>2)layers.7.self_attn.qkv_proj.q_proj.weight1
|
||||
output85<04><10>2layers.7.mlp.down_proj.weight/
|
||||
output86<04><0C>2layers.7.mlp.up_proj.weight:
|
||||
output87<02>2(layers.7.post_attention_layernorm.weight1
|
||||
output88<04><0C>2layers.7.mlp.gate_proj.weight;
|
||||
output89<04><10>2'layers.8.self_attn.o_proj.o_proj.weight=
|
||||
output90<04><02>2)layers.8.self_attn.qkv_proj.v_proj.weight1
|
||||
output91<02>2layers.8.input_layernorm.weight7
|
||||
output92<02>2%layers.8.self_attn.k_layernorm.weight=
|
||||
output93<04><02>2)layers.8.self_attn.qkv_proj.k_proj.weight7
|
||||
output94<02>2%layers.8.self_attn.q_layernorm.weight=
|
||||
output95<04><04>2)layers.8.self_attn.qkv_proj.q_proj.weight1
|
||||
output96<04><10>2layers.8.mlp.down_proj.weight/
|
||||
output97<04><0C>2layers.8.mlp.up_proj.weight:
|
||||
output98<02>2(layers.8.post_attention_layernorm.weight1
|
||||
output99<04><0C>2layers.8.mlp.gate_proj.weight<
|
||||
output100<04><10>2'layers.9.self_attn.o_proj.o_proj.weight>
|
||||
output101<04><02>2)layers.9.self_attn.qkv_proj.v_proj.weight2
|
||||
output102<02>2layers.9.input_layernorm.weight8
|
||||
output103<02>2%layers.9.self_attn.k_layernorm.weight>
|
||||
output104<04><02>2)layers.9.self_attn.qkv_proj.k_proj.weight8
|
||||
output105<02>2%layers.9.self_attn.q_layernorm.weight>
|
||||
output106<04><04>2)layers.9.self_attn.qkv_proj.q_proj.weight2
|
||||
output107<04><10>2layers.9.mlp.down_proj.weight0
|
||||
output108<04><0C>2layers.9.mlp.up_proj.weight;
|
||||
output109<02>2(layers.9.post_attention_layernorm.weight2
|
||||
output110<04><0C>2layers.9.mlp.gate_proj.weight=
|
||||
output111<04><10>2(layers.10.self_attn.o_proj.o_proj.weight?
|
||||
output112<04><02>2*layers.10.self_attn.qkv_proj.v_proj.weight3
|
||||
output113<02>2 layers.10.input_layernorm.weight9
|
||||
output114<02>2&layers.10.self_attn.k_layernorm.weight?
|
||||
output115<04><02>2*layers.10.self_attn.qkv_proj.k_proj.weight9
|
||||
output116<02>2&layers.10.self_attn.q_layernorm.weight?
|
||||
output117<04><04>2*layers.10.self_attn.qkv_proj.q_proj.weight3
|
||||
output118<04><10>2layers.10.mlp.down_proj.weight1
|
||||
output119<04><0C>2layers.10.mlp.up_proj.weight<
|
||||
output120<02>2)layers.10.post_attention_layernorm.weight3
|
||||
output121<04><0C>2layers.10.mlp.gate_proj.weight=
|
||||
output122<04><10>2(layers.11.self_attn.o_proj.o_proj.weight?
|
||||
output123<04><02>2*layers.11.self_attn.qkv_proj.v_proj.weight3
|
||||
output124<02>2 layers.11.input_layernorm.weight9
|
||||
output125<02>2&layers.11.self_attn.k_layernorm.weight?
|
||||
output126<04><02>2*layers.11.self_attn.qkv_proj.k_proj.weight9
|
||||
output127<02>2&layers.11.self_attn.q_layernorm.weight?
|
||||
output128<04><04>2*layers.11.self_attn.qkv_proj.q_proj.weight3
|
||||
output129<04><10>2layers.11.mlp.down_proj.weight1
|
||||
output130<04><0C>2layers.11.mlp.up_proj.weight<
|
||||
output131<02>2)layers.11.post_attention_layernorm.weight3
|
||||
output132<04><0C>2layers.11.mlp.gate_proj.weight=
|
||||
output133<04><10>2(layers.12.self_attn.o_proj.o_proj.weight?
|
||||
output134<04><02>2*layers.12.self_attn.qkv_proj.v_proj.weight3
|
||||
output135<02>2 layers.12.input_layernorm.weight9
|
||||
output136<02>2&layers.12.self_attn.k_layernorm.weight?
|
||||
output137<04><02>2*layers.12.self_attn.qkv_proj.k_proj.weight9
|
||||
output138<02>2&layers.12.self_attn.q_layernorm.weight?
|
||||
output139<04><04>2*layers.12.self_attn.qkv_proj.q_proj.weight3
|
||||
output140<04><10>2layers.12.mlp.down_proj.weight1
|
||||
output141<04><0C>2layers.12.mlp.up_proj.weight<
|
||||
output142<02>2)layers.12.post_attention_layernorm.weight3
|
||||
output143<04><0C>2layers.12.mlp.gate_proj.weight=
|
||||
output144<04><10>2(layers.13.self_attn.o_proj.o_proj.weight?
|
||||
output145<04><02>2*layers.13.self_attn.qkv_proj.v_proj.weight3
|
||||
output146<02>2 layers.13.input_layernorm.weight9
|
||||
output147<02>2&layers.13.self_attn.k_layernorm.weight?
|
||||
output148<04><02>2*layers.13.self_attn.qkv_proj.k_proj.weight9
|
||||
output149<02>2&layers.13.self_attn.q_layernorm.weight?
|
||||
output150<04><04>2*layers.13.self_attn.qkv_proj.q_proj.weight3
|
||||
output151<04><10>2layers.13.mlp.down_proj.weight1
|
||||
output152<04><0C>2layers.13.mlp.up_proj.weight<
|
||||
output153<02>2)layers.13.post_attention_layernorm.weight3
|
||||
output154<04><0C>2layers.13.mlp.gate_proj.weight=
|
||||
output155<04><10>2(layers.14.self_attn.o_proj.o_proj.weight?
|
||||
output156<04><02>2*layers.14.self_attn.qkv_proj.v_proj.weight3
|
||||
output157<02>2 layers.14.input_layernorm.weight9
|
||||
output158<02>2&layers.14.self_attn.k_layernorm.weight?
|
||||
output159<04><02>2*layers.14.self_attn.qkv_proj.k_proj.weight9
|
||||
output160<02>2&layers.14.self_attn.q_layernorm.weight?
|
||||
output161<04><04>2*layers.14.self_attn.qkv_proj.q_proj.weight3
|
||||
output162<04><10>2layers.14.mlp.down_proj.weight1
|
||||
output163<04><0C>2layers.14.mlp.up_proj.weight<
|
||||
output164<02>2)layers.14.post_attention_layernorm.weight3
|
||||
output165<04><0C>2layers.14.mlp.gate_proj.weight=
|
||||
output166<04><10>2(layers.15.self_attn.o_proj.o_proj.weight?
|
||||
output167<04><02>2*layers.15.self_attn.qkv_proj.v_proj.weight3
|
||||
output168<02>2 layers.15.input_layernorm.weight9
|
||||
output169<02>2&layers.15.self_attn.k_layernorm.weight?
|
||||
output170<04><02>2*layers.15.self_attn.qkv_proj.k_proj.weight9
|
||||
output171<02>2&layers.15.self_attn.q_layernorm.weight?
|
||||
output172<04><04>2*layers.15.self_attn.qkv_proj.q_proj.weight3
|
||||
output173<04><10>2layers.15.mlp.down_proj.weight1
|
||||
output174<04><0C>2layers.15.mlp.up_proj.weight<
|
||||
output175<02>2)layers.15.post_attention_layernorm.weight3
|
||||
output176<04><0C>2layers.15.mlp.gate_proj.weight=
|
||||
output177<04><10>2(layers.16.self_attn.o_proj.o_proj.weight?
|
||||
output178<04><02>2*layers.16.self_attn.qkv_proj.v_proj.weight3
|
||||
output179<02>2 layers.16.input_layernorm.weight9
|
||||
output180<02>2&layers.16.self_attn.k_layernorm.weight?
|
||||
output181<04><02>2*layers.16.self_attn.qkv_proj.k_proj.weight9
|
||||
output182<02>2&layers.16.self_attn.q_layernorm.weight?
|
||||
output183<04><04>2*layers.16.self_attn.qkv_proj.q_proj.weight3
|
||||
output184<04><10>2layers.16.mlp.down_proj.weight1
|
||||
output185<04><0C>2layers.16.mlp.up_proj.weight<
|
||||
output186<02>2)layers.16.post_attention_layernorm.weight3
|
||||
output187<04><0C>2layers.16.mlp.gate_proj.weight=
|
||||
output188<04><10>2(layers.17.self_attn.o_proj.o_proj.weight?
|
||||
output189<04><02>2*layers.17.self_attn.qkv_proj.v_proj.weight3
|
||||
output190<02>2 layers.17.input_layernorm.weight9
|
||||
output191<02>2&layers.17.self_attn.k_layernorm.weight?
|
||||
output192<04><02>2*layers.17.self_attn.qkv_proj.k_proj.weight9
|
||||
output193<02>2&layers.17.self_attn.q_layernorm.weight?
|
||||
output194<04><04>2*layers.17.self_attn.qkv_proj.q_proj.weight3
|
||||
output195<04><10>2layers.17.mlp.down_proj.weight1
|
||||
output196<04><0C>2layers.17.mlp.up_proj.weight<
|
||||
output197<02>2)layers.17.post_attention_layernorm.weight3
|
||||
output198<04><0C>2layers.17.mlp.gate_proj.weight=
|
||||
output199<04><10>2(layers.18.self_attn.o_proj.o_proj.weight?
|
||||
output200<04><02>2*layers.18.self_attn.qkv_proj.v_proj.weight3
|
||||
output201<02>2 layers.18.input_layernorm.weight9
|
||||
output202<02>2&layers.18.self_attn.k_layernorm.weight?
|
||||
output203<04><02>2*layers.18.self_attn.qkv_proj.k_proj.weight9
|
||||
output204<02>2&layers.18.self_attn.q_layernorm.weight?
|
||||
output205<04><04>2*layers.18.self_attn.qkv_proj.q_proj.weight3
|
||||
output206<04><10>2layers.18.mlp.down_proj.weight1
|
||||
output207<04><0C>2layers.18.mlp.up_proj.weight<
|
||||
output208<02>2)layers.18.post_attention_layernorm.weight3
|
||||
output209<04><0C>2layers.18.mlp.gate_proj.weight=
|
||||
output210<04><10>2(layers.19.self_attn.o_proj.o_proj.weight?
|
||||
output211<04><02>2*layers.19.self_attn.qkv_proj.v_proj.weight3
|
||||
output212<02>2 layers.19.input_layernorm.weight9
|
||||
output213<02>2&layers.19.self_attn.k_layernorm.weight?
|
||||
output214<04><02>2*layers.19.self_attn.qkv_proj.k_proj.weight9
|
||||
output215<02>2&layers.19.self_attn.q_layernorm.weight?
|
||||
output216<04><04>2*layers.19.self_attn.qkv_proj.q_proj.weight3
|
||||
output217<04><10>2layers.19.mlp.down_proj.weight1
|
||||
output218<04><0C>2layers.19.mlp.up_proj.weight<
|
||||
output219<02>2)layers.19.post_attention_layernorm.weight3
|
||||
output220<04><0C>2layers.19.mlp.gate_proj.weight=
|
||||
output221<04><10>2(layers.20.self_attn.o_proj.o_proj.weight?
|
||||
output222<04><02>2*layers.20.self_attn.qkv_proj.v_proj.weight3
|
||||
output223<02>2 layers.20.input_layernorm.weight9
|
||||
output224<02>2&layers.20.self_attn.k_layernorm.weight?
|
||||
output225<04><02>2*layers.20.self_attn.qkv_proj.k_proj.weight9
|
||||
output226<02>2&layers.20.self_attn.q_layernorm.weight?
|
||||
output227<04><04>2*layers.20.self_attn.qkv_proj.q_proj.weight3
|
||||
output228<04><10>2layers.20.mlp.down_proj.weight1
|
||||
output229<04><0C>2layers.20.mlp.up_proj.weight<
|
||||
output230<02>2)layers.20.post_attention_layernorm.weight3
|
||||
output231<04><0C>2layers.20.mlp.gate_proj.weight=
|
||||
output232<04><10>2(layers.21.self_attn.o_proj.o_proj.weight?
|
||||
output233<04><02>2*layers.21.self_attn.qkv_proj.v_proj.weight3
|
||||
output234<02>2 layers.21.input_layernorm.weight9
|
||||
output235<02>2&layers.21.self_attn.k_layernorm.weight?
|
||||
output236<04><02>2*layers.21.self_attn.qkv_proj.k_proj.weight9
|
||||
output237<02>2&layers.21.self_attn.q_layernorm.weight?
|
||||
output238<04><04>2*layers.21.self_attn.qkv_proj.q_proj.weight3
|
||||
output239<04><10>2layers.21.mlp.down_proj.weight1
|
||||
output240<04><0C>2layers.21.mlp.up_proj.weight<
|
||||
output241<02>2)layers.21.post_attention_layernorm.weight3
|
||||
output242<04><0C>2layers.21.mlp.gate_proj.weight=
|
||||
output243<04><10>2(layers.22.self_attn.o_proj.o_proj.weight?
|
||||
output244<04><02>2*layers.22.self_attn.qkv_proj.v_proj.weight3
|
||||
output245<02>2 layers.22.input_layernorm.weight9
|
||||
output246<02>2&layers.22.self_attn.k_layernorm.weight?
|
||||
output247<04><02>2*layers.22.self_attn.qkv_proj.k_proj.weight9
|
||||
output248<02>2&layers.22.self_attn.q_layernorm.weight?
|
||||
output249<04><04>2*layers.22.self_attn.qkv_proj.q_proj.weight3
|
||||
output250<04><10>2layers.22.mlp.down_proj.weight1
|
||||
output251<04><0C>2layers.22.mlp.up_proj.weight<
|
||||
output252<02>2)layers.22.post_attention_layernorm.weight3
|
||||
output253<04><0C>2layers.22.mlp.gate_proj.weight=
|
||||
output254<04><10>2(layers.23.self_attn.o_proj.o_proj.weight?
|
||||
output255<04><02>2*layers.23.self_attn.qkv_proj.v_proj.weight3
|
||||
output256<02>2 layers.23.input_layernorm.weight9
|
||||
output257<02>2&layers.23.self_attn.k_layernorm.weight?
|
||||
output258<04><02>2*layers.23.self_attn.qkv_proj.k_proj.weight9
|
||||
output259<02>2&layers.23.self_attn.q_layernorm.weight?
|
||||
output260<04><04>2*layers.23.self_attn.qkv_proj.q_proj.weight3
|
||||
output261<04><10>2layers.23.mlp.down_proj.weight1
|
||||
output262<04><0C>2layers.23.mlp.up_proj.weight<
|
||||
output263<02>2)layers.23.post_attention_layernorm.weight3
|
||||
output264<04><0C>2layers.23.mlp.gate_proj.weight=
|
||||
output265<04><10>2(layers.24.self_attn.o_proj.o_proj.weight?
|
||||
output266<04><02>2*layers.24.self_attn.qkv_proj.v_proj.weight3
|
||||
output267<02>2 layers.24.input_layernorm.weight9
|
||||
output268<02>2&layers.24.self_attn.k_layernorm.weight?
|
||||
output269<04><02>2*layers.24.self_attn.qkv_proj.k_proj.weight9
|
||||
output270<02>2&layers.24.self_attn.q_layernorm.weight?
|
||||
output271<04><04>2*layers.24.self_attn.qkv_proj.q_proj.weight3
|
||||
output272<04><10>2layers.24.mlp.down_proj.weight1
|
||||
output273<04><0C>2layers.24.mlp.up_proj.weight<
|
||||
output274<02>2)layers.24.post_attention_layernorm.weight3
|
||||
output275<04><0C>2layers.24.mlp.gate_proj.weight=
|
||||
output276<04><10>2(layers.25.self_attn.o_proj.o_proj.weight?
|
||||
output277<04><02>2*layers.25.self_attn.qkv_proj.v_proj.weight3
|
||||
output278<02>2 layers.25.input_layernorm.weight9
|
||||
output279<02>2&layers.25.self_attn.k_layernorm.weight?
|
||||
output280<04><02>2*layers.25.self_attn.qkv_proj.k_proj.weight9
|
||||
output281<02>2&layers.25.self_attn.q_layernorm.weight?
|
||||
output282<04><04>2*layers.25.self_attn.qkv_proj.q_proj.weight3
|
||||
output283<04><10>2layers.25.mlp.down_proj.weight1
|
||||
output284<04><0C>2layers.25.mlp.up_proj.weight<
|
||||
output285<02>2)layers.25.post_attention_layernorm.weight3
|
||||
output286<04><0C>2layers.25.mlp.gate_proj.weight=
|
||||
output287<04><10>2(layers.26.self_attn.o_proj.o_proj.weight?
|
||||
output288<04><02>2*layers.26.self_attn.qkv_proj.v_proj.weight3
|
||||
output289<02>2 layers.26.input_layernorm.weight9
|
||||
output290<02>2&layers.26.self_attn.k_layernorm.weight?
|
||||
output291<04><02>2*layers.26.self_attn.qkv_proj.k_proj.weight9
|
||||
output292<02>2&layers.26.self_attn.q_layernorm.weight?
|
||||
output293<04><04>2*layers.26.self_attn.qkv_proj.q_proj.weight3
|
||||
output294<04><10>2layers.26.mlp.down_proj.weight1
|
||||
output295<04><0C>2layers.26.mlp.up_proj.weight<
|
||||
output296<02>2)layers.26.post_attention_layernorm.weight3
|
||||
output297<04><0C>2layers.26.mlp.gate_proj.weight=
|
||||
output298<04><10>2(layers.27.self_attn.o_proj.o_proj.weight?
|
||||
output299<04><02>2*layers.27.self_attn.qkv_proj.v_proj.weight3
|
||||
output300<02>2 layers.27.input_layernorm.weight9
|
||||
output301<02>2&layers.27.self_attn.k_layernorm.weight?
|
||||
output302<04><02>2*layers.27.self_attn.qkv_proj.k_proj.weight9
|
||||
output303<02>2&layers.27.self_attn.q_layernorm.weight?
|
||||
output304<04><04>2*layers.27.self_attn.qkv_proj.q_proj.weight3
|
||||
output305<04><10>2layers.27.mlp.down_proj.weight1
|
||||
output306<04><0C>2layers.27.mlp.up_proj.weight<
|
||||
output307<02>2)layers.27.post_attention_layernorm.weight3
|
||||
output308<04><0C>2layers.27.mlp.gate_proj.weight$
|
||||
output309<05><><02>2lm_head.weight
|
||||
output310<02>2norm.weight
|
||||
3
layout_opt/model/graph.hlo
Normal file
3
layout_opt/model/graph.hlo
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b26580ab7144484c2a38bd00978d0df769574a4d80bac429f0f4b37ce9e628b4
|
||||
size 196610
|
||||
151388
merges.txt
Normal file
151388
merges.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
model-00001-of-00002.safetensors
Normal file
3
model-00001-of-00002.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:169ad53ec313c3a34b06c0809216e4fc072cce444a5d4ff2b59690d064130ed5
|
||||
size 3441185608
|
||||
3
model-00002-of-00002.safetensors
Normal file
3
model-00002-of-00002.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:912becff8d60672aa8628ef08c05898d9adf17c2ad4ae3caf99b065622fdeff9
|
||||
size 622329984
|
||||
3
model.pt
Normal file
3
model.pt
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:312d7cdb71a48e347a18a236be5a96c47f1e7f2075c920123708df9b2c06bedf
|
||||
size 47333643
|
||||
318
model.safetensors.index.json
Normal file
318
model.safetensors.index.json
Normal file
@@ -0,0 +1,318 @@
|
||||
{
|
||||
"metadata": {
|
||||
"total_size": 4063479808
|
||||
},
|
||||
"weight_map": {
|
||||
"lm_head.weight": "model-00002-of-00002.safetensors",
|
||||
"model.embed_tokens.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.norm.weight": "model-00001-of-00002.safetensors"
|
||||
}
|
||||
}
|
||||
222
neuron_config.json
Normal file
222
neuron_config.json
Normal file
@@ -0,0 +1,222 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 4,
|
||||
"bucket_n_active_tokens": false,
|
||||
"buckets": [
|
||||
2048
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 2,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": null,
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": null,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 2048,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": null,
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": null,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
token_generation_model/_tp0_bk0/command.txt
Normal file
1
token_generation_model/_tp0_bk0/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_5e3d0ce8963512f9ea2d+937cd7a2.hlo_module.pb --output model.MODULE_5e3d0ce8963512f9ea2d+937cd7a2.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --enable-internal-neff-wrapper --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
|
||||
590
token_generation_model/_tp0_bk0/global_metric_store.json
Normal file
590
token_generation_model/_tp0_bk0/global_metric_store.json
Normal file
@@ -0,0 +1,590 @@
|
||||
{
|
||||
"Average": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 96.97119140625,
|
||||
"StaticProfiler::AveragePartitionUtilization": 88.53791809082031,
|
||||
"StaticProfiler::AveragePeUtilization": 81.30671691894531,
|
||||
"StaticProfiler::LocalizationEfficiency": 161.8649139404297,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 167.2097930908203,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
|
||||
}
|
||||
},
|
||||
"Count": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 1,
|
||||
"StaticProfiler::AveragePartitionUtilization": 1,
|
||||
"StaticProfiler::AveragePeUtilization": 1,
|
||||
"StaticProfiler::LocalizationEfficiency": 1,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
|
||||
}
|
||||
},
|
||||
"Sum": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 2.377948760986328,
|
||||
"AffinePredicateResolution": 0.03525829315185547,
|
||||
"AliasDependencyElimination": 0.0024094581604003906,
|
||||
"AliasDependencyInduction": 0.2979590892791748,
|
||||
"AliasDependencyReset": 0.30464816093444824,
|
||||
"BFComputeCutting": 0.07316970825195313,
|
||||
"BirCodeGenLoop": 2.482311487197876,
|
||||
"CCOpFusion": 0.5325038433074951,
|
||||
"CanonicalizeConv": 0.00023499999952036887,
|
||||
"CanonicalizeDAGForPGTiling": 0.1513075828552246,
|
||||
"CanonicalizeForTensorizer": 0.00026500000967644155,
|
||||
"CanonicalizeIR": 0.04851794242858887,
|
||||
"Canonicalizer": 0.004513999912887812,
|
||||
"CoalesceCCOp": 0.16019058227539063,
|
||||
"CommuteConcat": 0.024104833602905273,
|
||||
"DMALocalityOpt": 0.037362098693847656,
|
||||
"DMAProfiler": 0.07356810569763184,
|
||||
"DMATilingProfiler": 0.09021782875061035,
|
||||
"DataLocalityOpt": 3.034395694732666,
|
||||
"DataStreaming": 0.1286299228668213,
|
||||
"DeConcat": 0.04242873191833496,
|
||||
"DeadCodeElimination": 0.02465653419494629,
|
||||
"DeadStoreElimination": 0.8050529956817627,
|
||||
"DelinearIndices": 0.46034955978393555,
|
||||
"Delinearization": 0.11224937438964844,
|
||||
"DelinearizeSPMD": 0.14354729652404785,
|
||||
"DoNothing": 0.00033402442932128906,
|
||||
"DramToDramTranspose": 0.2800898551940918,
|
||||
"DumpGraphAndMetadata": 0.14458084106445313,
|
||||
"EliminateDivs": 0.11128425598144531,
|
||||
"ExpandBatchNorm": 0.04912686347961426,
|
||||
"ExpandISAMacro": 0.07862186431884766,
|
||||
"FactorizeBlkDims": 0.49788737297058105,
|
||||
"FactorizeThreadAxesInFreeDims": 0.05206179618835449,
|
||||
"FlattenMacroLoop": 0.08013129234313965,
|
||||
"GenericAccessSimplifier": 0.022362232208251953,
|
||||
"HoistCompute": 3.899999865097925e-05,
|
||||
"IdentifyCrossPassTensors": 0.00022899999748915434,
|
||||
"InferInitValue": 1.3044743537902832,
|
||||
"InferIntrinsicOnCC": 0.26564502716064453,
|
||||
"InferNeuronTensor": 1.4273626804351807,
|
||||
"InferNonlocalTensors": 3.070617198944092,
|
||||
"InferPSumTensor": 1.877610206604004,
|
||||
"InferShardAxis": 4.046135902404785,
|
||||
"InferSharedMemLoc": 0.10418868064880371,
|
||||
"InlineNativeKernels": 0.046288251876831055,
|
||||
"InsertCoreBarrier": 0.12113451957702637,
|
||||
"InsertIOTransposes": 1.5809462070465088,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.37781310081481934,
|
||||
"InsertLocalTransposes": 0.8543226718902588,
|
||||
"InsertOffloadedTransposes": 0.0779869556427002,
|
||||
"LICM": 0.10844612121582031,
|
||||
"LateLegalizeInst": 0.1468517780303955,
|
||||
"LateLegalizePostSplit": 0.08665323257446289,
|
||||
"LateLowerReshapeOp": 0.02942204475402832,
|
||||
"LateLowerTensorOp": 0.22415471076965332,
|
||||
"LateNeuronInstComb": 0.9669840335845947,
|
||||
"LayoutPreprocessing": 0.7032866477966309,
|
||||
"LayoutPreprocessingAndAnalysis": 1.1142585277557373,
|
||||
"LayoutRequirementAnalysis": 0.40787601470947266,
|
||||
"LegalizeCCOpLayout": 0.052059173583984375,
|
||||
"LegalizeOpLevelAlias": 0.022985219955444336,
|
||||
"LegalizePartitionReduce": 0.036779165267944336,
|
||||
"LegalizeSundaAccess": 0.9409115314483643,
|
||||
"LegalizeSundaMacro": 0.7163646221160889,
|
||||
"LegalizeType": 0.1376960277557373,
|
||||
"LocalLayoutOpt": 0.5141463279724121,
|
||||
"LoopFusion": 0.2369997501373291,
|
||||
"LoopSplitting": 0.026223182678222656,
|
||||
"LowerBroadcast": 0.04980111122131348,
|
||||
"LowerCCOpBlockAxis": 0.17850804328918457,
|
||||
"LowerComplexBroadcast": 0.06136465072631836,
|
||||
"LowerIntrinsics": 0.9023723602294922,
|
||||
"LowerShardAxis": 0.19546747207641602,
|
||||
"LowerTensorOp": 0.38861823081970215,
|
||||
"LowerToSendRecv": 0.1479358673095703,
|
||||
"LowerTranspose": 0.41634607315063477,
|
||||
"MacroGeneration": 1.907278060913086,
|
||||
"MaskPropagation": 0.10989689826965332,
|
||||
"MemcastMotion": 0.00011899999663000926,
|
||||
"MemcpyElimination": 3.3127715587615967,
|
||||
"MutateDataType": 0.031205177307128906,
|
||||
"NeuronAliasDependencyInduction": 0.014397859573364258,
|
||||
"NeuronAliasDependencyReset": 0.018595218658447266,
|
||||
"NeuronInstComb": 0.29677605628967285,
|
||||
"NeuronLICM": 0.2591841220855713,
|
||||
"NeuronLoopFusion": 0.9968361854553223,
|
||||
"NeuronLoopInterchange": 0.049851179122924805,
|
||||
"NeuronSimplifier": 0.4482152462005615,
|
||||
"NeuronSimplifyPredicates": 0.13344645500183105,
|
||||
"NeuronValueNumbering": 0.10327720642089844,
|
||||
"OptimizeAliasedCopyChain": 0.010796785354614258,
|
||||
"OptimizeNKIKernels": 0.9786763191223145,
|
||||
"PAGLayoutOpt": 11.095458984375,
|
||||
"PComputeCutting": 0.28275370597839355,
|
||||
"PGLayoutTilingPipeline": 27.67894172668457,
|
||||
"PGTiling": 4.957591533660889,
|
||||
"PadElimination": 0.009732246398925781,
|
||||
"ParAxesAnnotation": 10.236833572387695,
|
||||
"PartialLoopFusion": 1.0972201824188232,
|
||||
"PartialSimdFusion": 0.529569149017334,
|
||||
"PenguinizeFunctions": 0.00012799999967683107,
|
||||
"PerfectLoopNest": 0.059952735900878906,
|
||||
"PruneFunctions": 0.000391999987186864,
|
||||
"RecognizeOpIdiom": 0.120391845703125,
|
||||
"Recompute": 0.005687713623046875,
|
||||
"RelaxPredicates": 0.09513998031616211,
|
||||
"Rematerialization": 0.14099407196044922,
|
||||
"RemoveOptimizationBarriers": 0.0004239999980200082,
|
||||
"RemoveShardedPartitionAxes": 0.6145329475402832,
|
||||
"ReshapeWeights": 0.0214080810546875,
|
||||
"ResolveAccessConflict": 0.1656482219696045,
|
||||
"ResolveComplicatePredicates": 0.03499007225036621,
|
||||
"RewriteReplicationMatmul": 0.03796839714050293,
|
||||
"RewriteWeights": 0.06430387496948242,
|
||||
"SFKVectorizer": 5.783308029174805,
|
||||
"ScatterMotion": 0.002827000105753541,
|
||||
"ShardingPropagationAnalysis": 0.5885381698608398,
|
||||
"SimpleAllReduceTiling": 0.05830836296081543,
|
||||
"Simplifier": 0.07847452163696289,
|
||||
"SimplifyMacroPredicates": 0.25547003746032715,
|
||||
"SimplifyNeuronTensor": 0.3350214958190918,
|
||||
"SimplifySlice": 0.02300572395324707,
|
||||
"SimplifyTensor": 0.2370157241821289,
|
||||
"SpillPSum": 0.4782223701477051,
|
||||
"SplitAPUnionSets": 0.6796882152557373,
|
||||
"SplitAccGrp": 0.03665971755981445,
|
||||
"StaticProfiler": 0.13864398002624512,
|
||||
"StaticTransposeLocalTensor": 0.2623753547668457,
|
||||
"SundaISel": 1.3397884368896484,
|
||||
"TCTransform": 0.025269269943237305,
|
||||
"TensorInitialization": 0.16660237312316895,
|
||||
"TensorOpSimplifier": 0.15259718894958496,
|
||||
"TensorOpTransform": 0.8282039165496826,
|
||||
"TensorizerLegalizationPass": 0.00011800000356743112,
|
||||
"TileCCOps": 0.3066561222076416,
|
||||
"TilingProfiler": 0.33243393898010254,
|
||||
"TransformConvOp": 0.05726790428161621,
|
||||
"TritiumFusion": 0.13151931762695313,
|
||||
"ValueNumbering": 0.06717801094055176,
|
||||
"VectorizeDMA": 0.5187788009643555,
|
||||
"VectorizeMatMult": 0.04601097106933594,
|
||||
"VerifySupportedOps": 0.000195999993593432,
|
||||
"WeightCoalescing": 0.05271005630493164,
|
||||
"ZeroSizeTensorElimination": 0.0003685951232910156,
|
||||
"algsimp": 0.0011439999798312783,
|
||||
"batchnorm_expander": 0.0005740000051446259,
|
||||
"boundary-marker-removal": 0.00017699999443721026,
|
||||
"call-inliner": 0.00013800000306218863,
|
||||
"canonicalize-boundary-marker": 0.0003129999968223274,
|
||||
"collective-stream-id-checker": 7.899999764049426e-05,
|
||||
"comparison-expander": 0.000195999993593432,
|
||||
"computation-deduplicator": 0.00031900001340545714,
|
||||
"config-lowering": 0.00015100000018719584,
|
||||
"constant_folding": 9.40000027185306e-05,
|
||||
"cse": 0.00038400001358240843,
|
||||
"dce": 2.4000000848900527e-05,
|
||||
"dynamic-slice-transpose": 8.099999831756577e-05,
|
||||
"eliminate-redundant-compare": 7.999999797903001e-05,
|
||||
"emit-offloaded-dropout": 0.00013899999612476677,
|
||||
"flatten-call-graph": 0.0001720000000204891,
|
||||
"fuse-send-recv": 0.0008440000237897038,
|
||||
"hilo-conditional-to-select": 5.2999999752501026e-05,
|
||||
"hilo::LegalizeAlias": 0.0019950000569224358,
|
||||
"hilo::NeuronInstCombine": 0.0008609999786131084,
|
||||
"hilo::NeuronOpFusion": 0.0003239999932702631,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.00031400000443682075,
|
||||
"hilo::ScheduleFusion": 2.9000000722589903e-05,
|
||||
"hilo::SixtyFourHack": 0.00023200000578071922,
|
||||
"hilo::VerifyAliasing": 5.0999999075429514e-05,
|
||||
"hlo-mac-count": 0.006806999910622835,
|
||||
"io-con-pipe-begin": 9.999999747378752e-06,
|
||||
"io-con-pipe-end": 0.0,
|
||||
"io-layout-normalization": 0.0014359999913722277,
|
||||
"legalize-ccops-for-tensorizer": 1.4999999621068127e-05,
|
||||
"legalize-compare": 0.000155999994603917,
|
||||
"lower-argminmax-custom-call": 7.599999662488699e-05,
|
||||
"map-inline": 0.0003760000108741224,
|
||||
"metadata-naming": 0.0007949999999254942,
|
||||
"mlir::detail::OpToOpPassAdaptor": 0.00014600000577047467,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.057739999145269394,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.0017610000213608146,
|
||||
"mlir::mhlo::LowerComplexPass": 0.00240899994969368,
|
||||
"native-to-custom-softmax": 0.00022000000171829015,
|
||||
"native-to-custom-softmax-dx": 0.00023200000578071922,
|
||||
"neuron-hlo-verifier": 0.01620600000023842,
|
||||
"operand_upcaster": 0.0005629999795928597,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 0.0,
|
||||
"post-partition-simplification": 0.03471200168132782,
|
||||
"pre-hlo-begin": 3.000000106112566e-06,
|
||||
"pre-hlo-end": 0.0,
|
||||
"replace-minimum-constant": 0.00010599999950500205,
|
||||
"reshape-mover": 4.5000000682193786e-05,
|
||||
"simplify-concat": 0.0008580000139772892,
|
||||
"simplify-while-loops": 3.300000025774352e-05,
|
||||
"transform-variadic-reduce": 0.00025599999935366213,
|
||||
"tuple-simplifier": 8.900000102585182e-05,
|
||||
"unpack-nested-aws-ntwsr": 0.00014000000373926014,
|
||||
"unroll-while-loop": 6.000000212225132e-06
|
||||
},
|
||||
"hilo": {
|
||||
"HloMacCount": 1751965696.0,
|
||||
"Traffic": 1252616320.0
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 41972,
|
||||
"StaticProfiler::AifUb": 17.11549949645996,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 27.70398712158203,
|
||||
"StaticProfiler::AverageDmaLength": 2723.7119140625,
|
||||
"StaticProfiler::DDRTransferBytes": 918042224,
|
||||
"StaticProfiler::InternalTransferBytes": 174444896,
|
||||
"StaticProfiler::LoadExpanded": 238040,
|
||||
"StaticProfiler::StoreExpanded": 18609,
|
||||
"StaticProfiler::TotalDMAExpanded": 256649,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 57730,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50535,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 179,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 30304,
|
||||
"TilingProfiler::NumPfTransposes": 348,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 198,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 7835,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 5217,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 508,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 2110,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 59,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2266,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"all": {
|
||||
"compiletime": {
|
||||
"CanonicalizeConv": 0.00023499999952036887,
|
||||
"CanonicalizeForTensorizer": 0.00026500000967644155,
|
||||
"Canonicalizer": 0.004513999912887812,
|
||||
"HoistCompute": 3.899999865097925e-05,
|
||||
"IdentifyCrossPassTensors": 0.00022899999748915434,
|
||||
"MemcastMotion": 0.00011899999663000926,
|
||||
"PenguinizeFunctions": 0.00012799999967683107,
|
||||
"PruneFunctions": 0.000391999987186864,
|
||||
"RemoveOptimizationBarriers": 0.0004239999980200082,
|
||||
"ScatterMotion": 0.002827000105753541,
|
||||
"TensorizerLegalizationPass": 0.00011800000356743112,
|
||||
"VerifySupportedOps": 0.000195999993593432,
|
||||
"algsimp": 0.0011439999798312783,
|
||||
"batchnorm_expander": 0.0005740000051446259,
|
||||
"boundary-marker-removal": 0.00017699999443721026,
|
||||
"call-inliner": 0.00013800000306218863,
|
||||
"canonicalize-boundary-marker": 0.0003129999968223274,
|
||||
"collective-stream-id-checker": 7.899999764049426e-05,
|
||||
"comparison-expander": 0.000195999993593432,
|
||||
"computation-deduplicator": 0.00031900001340545714,
|
||||
"config-lowering": 0.00015100000018719584,
|
||||
"constant_folding": 9.40000027185306e-05,
|
||||
"cse": 0.00038400001358240843,
|
||||
"dce": 2.4000000848900527e-05,
|
||||
"dynamic-slice-transpose": 8.099999831756577e-05,
|
||||
"eliminate-redundant-compare": 7.999999797903001e-05,
|
||||
"emit-offloaded-dropout": 0.00013899999612476677,
|
||||
"flatten-call-graph": 0.0001720000000204891,
|
||||
"fuse-send-recv": 0.0008440000237897038,
|
||||
"hilo-conditional-to-select": 5.2999999752501026e-05,
|
||||
"hilo::LegalizeAlias": 0.0019950000569224358,
|
||||
"hilo::NeuronInstCombine": 0.0008609999786131084,
|
||||
"hilo::NeuronOpFusion": 0.0003239999932702631,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.00031400000443682075,
|
||||
"hilo::ScheduleFusion": 2.9000000722589903e-05,
|
||||
"hilo::SixtyFourHack": 0.00023200000578071922,
|
||||
"hilo::VerifyAliasing": 5.0999999075429514e-05,
|
||||
"hlo-mac-count": 0.006806999910622835,
|
||||
"io-con-pipe-begin": 9.999999747378752e-06,
|
||||
"io-con-pipe-end": 0.0,
|
||||
"io-layout-normalization": 0.0014359999913722277,
|
||||
"legalize-ccops-for-tensorizer": 1.4999999621068127e-05,
|
||||
"legalize-compare": 0.000155999994603917,
|
||||
"lower-argminmax-custom-call": 7.599999662488699e-05,
|
||||
"map-inline": 0.0003760000108741224,
|
||||
"metadata-naming": 0.0007949999999254942,
|
||||
"mlir::detail::OpToOpPassAdaptor": 0.00014600000577047467,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.057739999145269394,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.0017610000213608146,
|
||||
"mlir::mhlo::LowerComplexPass": 0.00240899994969368,
|
||||
"native-to-custom-softmax": 0.00022000000171829015,
|
||||
"native-to-custom-softmax-dx": 0.00023200000578071922,
|
||||
"neuron-hlo-verifier": 0.01620600000023842,
|
||||
"operand_upcaster": 0.0005629999795928597,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 0.0,
|
||||
"post-partition-simplification": 0.03471200168132782,
|
||||
"pre-hlo-begin": 3.000000106112566e-06,
|
||||
"pre-hlo-end": 0.0,
|
||||
"replace-minimum-constant": 0.00010599999950500205,
|
||||
"reshape-mover": 4.5000000682193786e-05,
|
||||
"simplify-concat": 0.0008580000139772892,
|
||||
"simplify-while-loops": 3.300000025774352e-05,
|
||||
"transform-variadic-reduce": 0.00025599999935366213,
|
||||
"tuple-simplifier": 8.900000102585182e-05,
|
||||
"unpack-nested-aws-ntwsr": 0.00014000000373926014,
|
||||
"unroll-while-loop": 6.000000212225132e-06
|
||||
}
|
||||
},
|
||||
"cumsum": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.0002205371856689453,
|
||||
"DMALocalityOpt": 0.00016808509826660156,
|
||||
"DMAProfiler": 0.0006198883056640625,
|
||||
"DataStreaming": 0.0002627372741699219,
|
||||
"DoNothing": 0.00013875961303710938,
|
||||
"ExpandISAMacro": 0.0005130767822265625,
|
||||
"FactorizeBlkDims": 0.0005578994750976563,
|
||||
"InferPSumTensor": 0.0005154609680175781,
|
||||
"InferSharedMemLoc": 0.00028228759765625,
|
||||
"InsertCoreBarrier": 0.00024247169494628906,
|
||||
"LateLegalizeInst": 0.0003752708435058594,
|
||||
"LateNeuronInstComb": 0.0005624294281005859,
|
||||
"LegalizeSundaAccess": 0.0013544559478759766,
|
||||
"LegalizeType": 0.00024437904357910156,
|
||||
"LowerBroadcast": 0.00022649765014648438,
|
||||
"LowerIntrinsics": 0.00021767616271972656,
|
||||
"LowerTranspose": 0.0002415180206298828,
|
||||
"NeuronInstComb": 0.0007607936859130859,
|
||||
"NeuronLICM": 0.00041222572326660156,
|
||||
"NeuronSimplifyPredicates": 0.002053976058959961,
|
||||
"NeuronValueNumbering": 0.0004031658172607422,
|
||||
"SFKVectorizer": 0.002403736114501953,
|
||||
"SimpleAllReduceTiling": 0.0002071857452392578,
|
||||
"SimplifyNeuronTensor": 0.0004944801330566406,
|
||||
"SpillPSum": 0.0004985332489013672,
|
||||
"WeightCoalescing": 0.000213623046875
|
||||
}
|
||||
},
|
||||
"sg00": {
|
||||
"hilo": {
|
||||
"ArithmeticIntensity": 2.797290325164795,
|
||||
"HloMacCount": 1751965696.0,
|
||||
"Traffic": 1252616320.0
|
||||
}
|
||||
},
|
||||
"sg0000": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 2.377948760986328,
|
||||
"AffinePredicateResolution": 0.03525829315185547,
|
||||
"AliasDependencyElimination": 0.0024094581604003906,
|
||||
"AliasDependencyInduction": 0.2979590892791748,
|
||||
"AliasDependencyReset": 0.30464816093444824,
|
||||
"BFComputeCutting": 0.07316970825195313,
|
||||
"BirCodeGenLoop": 2.482311487197876,
|
||||
"CCOpFusion": 0.5325038433074951,
|
||||
"CanonicalizeDAGForPGTiling": 0.1513075828552246,
|
||||
"CanonicalizeIR": 0.04851794242858887,
|
||||
"CoalesceCCOp": 0.1563706398010254,
|
||||
"CommuteConcat": 0.024104833602905273,
|
||||
"DMALocalityOpt": 0.03418087959289551,
|
||||
"DMAProfiler": 0.06951689720153809,
|
||||
"DMATilingProfiler": 0.09021782875061035,
|
||||
"DataLocalityOpt": 3.034395694732666,
|
||||
"DataStreaming": 0.12200498580932617,
|
||||
"DeConcat": 0.04242873191833496,
|
||||
"DeadCodeElimination": 0.02465653419494629,
|
||||
"DeadStoreElimination": 0.8050529956817627,
|
||||
"DelinearIndices": 0.46034955978393555,
|
||||
"Delinearization": 0.11224937438964844,
|
||||
"DelinearizeSPMD": 0.14354729652404785,
|
||||
"DoNothing": 6.198883056640625e-05,
|
||||
"DramToDramTranspose": 0.2800898551940918,
|
||||
"DumpGraphAndMetadata": 0.14458084106445313,
|
||||
"EliminateDivs": 0.11128425598144531,
|
||||
"ExpandBatchNorm": 0.04912686347961426,
|
||||
"ExpandISAMacro": 0.07448863983154297,
|
||||
"FactorizeBlkDims": 0.48647499084472656,
|
||||
"FactorizeThreadAxesInFreeDims": 0.05206179618835449,
|
||||
"FlattenMacroLoop": 0.08013129234313965,
|
||||
"GenericAccessSimplifier": 0.022362232208251953,
|
||||
"InferInitValue": 1.3044743537902832,
|
||||
"InferIntrinsicOnCC": 0.26564502716064453,
|
||||
"InferNeuronTensor": 1.4273626804351807,
|
||||
"InferNonlocalTensors": 3.070617198944092,
|
||||
"InferPSumTensor": 1.8667116165161133,
|
||||
"InferShardAxis": 4.046135902404785,
|
||||
"InferSharedMemLoc": 0.10098028182983398,
|
||||
"InlineNativeKernels": 0.046288251876831055,
|
||||
"InsertCoreBarrier": 0.1175682544708252,
|
||||
"InsertIOTransposes": 1.5809462070465088,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.37781310081481934,
|
||||
"InsertLocalTransposes": 0.8543226718902588,
|
||||
"InsertOffloadedTransposes": 0.0779869556427002,
|
||||
"LICM": 0.10844612121582031,
|
||||
"LateLegalizeInst": 0.13918232917785645,
|
||||
"LateLegalizePostSplit": 0.08665323257446289,
|
||||
"LateLowerReshapeOp": 0.02942204475402832,
|
||||
"LateLowerTensorOp": 0.22415471076965332,
|
||||
"LateNeuronInstComb": 0.9580295085906982,
|
||||
"LayoutPreprocessing": 0.7032866477966309,
|
||||
"LayoutPreprocessingAndAnalysis": 1.1142585277557373,
|
||||
"LayoutRequirementAnalysis": 0.40787601470947266,
|
||||
"LegalizeCCOpLayout": 0.052059173583984375,
|
||||
"LegalizeOpLevelAlias": 0.022985219955444336,
|
||||
"LegalizePartitionReduce": 0.036779165267944336,
|
||||
"LegalizeSundaAccess": 0.9256985187530518,
|
||||
"LegalizeSundaMacro": 0.7163646221160889,
|
||||
"LegalizeType": 0.12823128700256348,
|
||||
"LocalLayoutOpt": 0.5141463279724121,
|
||||
"LoopFusion": 0.2369997501373291,
|
||||
"LoopSplitting": 0.026223182678222656,
|
||||
"LowerBroadcast": 0.04629969596862793,
|
||||
"LowerCCOpBlockAxis": 0.17850804328918457,
|
||||
"LowerComplexBroadcast": 0.06136465072631836,
|
||||
"LowerIntrinsics": 0.898535966873169,
|
||||
"LowerShardAxis": 0.19546747207641602,
|
||||
"LowerTensorOp": 0.38861823081970215,
|
||||
"LowerToSendRecv": 0.1479358673095703,
|
||||
"LowerTranspose": 0.4128274917602539,
|
||||
"MacroGeneration": 1.907278060913086,
|
||||
"MaskPropagation": 0.10989689826965332,
|
||||
"MemcpyElimination": 3.3127715587615967,
|
||||
"MutateDataType": 0.031205177307128906,
|
||||
"NeuronAliasDependencyInduction": 0.014397859573364258,
|
||||
"NeuronAliasDependencyReset": 0.018595218658447266,
|
||||
"NeuronInstComb": 0.28766393661499023,
|
||||
"NeuronLICM": 0.24966120719909668,
|
||||
"NeuronLoopFusion": 0.9968361854553223,
|
||||
"NeuronLoopInterchange": 0.049851179122924805,
|
||||
"NeuronSimplifier": 0.4482152462005615,
|
||||
"NeuronSimplifyPredicates": 0.12775635719299316,
|
||||
"NeuronValueNumbering": 0.09903788566589355,
|
||||
"OptimizeAliasedCopyChain": 0.010796785354614258,
|
||||
"OptimizeNKIKernels": 0.9786763191223145,
|
||||
"PAGLayoutOpt": 11.095458984375,
|
||||
"PComputeCutting": 0.28275370597839355,
|
||||
"PGLayoutTilingPipeline": 27.67894172668457,
|
||||
"PGTiling": 4.957591533660889,
|
||||
"PadElimination": 0.009732246398925781,
|
||||
"ParAxesAnnotation": 10.236833572387695,
|
||||
"PartialLoopFusion": 1.0972201824188232,
|
||||
"PartialSimdFusion": 0.529569149017334,
|
||||
"PerfectLoopNest": 0.059952735900878906,
|
||||
"RecognizeOpIdiom": 0.120391845703125,
|
||||
"Recompute": 0.005687713623046875,
|
||||
"RelaxPredicates": 0.09513998031616211,
|
||||
"Rematerialization": 0.14099407196044922,
|
||||
"RemoveShardedPartitionAxes": 0.6145329475402832,
|
||||
"ReshapeWeights": 0.0214080810546875,
|
||||
"ResolveAccessConflict": 0.1656482219696045,
|
||||
"ResolveComplicatePredicates": 0.03499007225036621,
|
||||
"RewriteReplicationMatmul": 0.03796839714050293,
|
||||
"RewriteWeights": 0.06430387496948242,
|
||||
"SFKVectorizer": 5.7498369216918945,
|
||||
"ShardingPropagationAnalysis": 0.5885381698608398,
|
||||
"SimpleAllReduceTiling": 0.05400824546813965,
|
||||
"Simplifier": 0.07847452163696289,
|
||||
"SimplifyMacroPredicates": 0.25547003746032715,
|
||||
"SimplifyNeuronTensor": 0.2766604423522949,
|
||||
"SimplifySlice": 0.02300572395324707,
|
||||
"SimplifyTensor": 0.2370157241821289,
|
||||
"SpillPSum": 0.4576537609100342,
|
||||
"SplitAPUnionSets": 0.6796882152557373,
|
||||
"SplitAccGrp": 0.03665971755981445,
|
||||
"StaticProfiler": 0.13864398002624512,
|
||||
"StaticTransposeLocalTensor": 0.2623753547668457,
|
||||
"SundaISel": 1.3397884368896484,
|
||||
"TCTransform": 0.025269269943237305,
|
||||
"TensorInitialization": 0.16660237312316895,
|
||||
"TensorOpSimplifier": 0.15259718894958496,
|
||||
"TensorOpTransform": 0.8282039165496826,
|
||||
"TileCCOps": 0.3066561222076416,
|
||||
"TilingProfiler": 0.33243393898010254,
|
||||
"TransformConvOp": 0.05726790428161621,
|
||||
"TritiumFusion": 0.13151931762695313,
|
||||
"ValueNumbering": 0.06717801094055176,
|
||||
"VectorizeDMA": 0.5187788009643555,
|
||||
"VectorizeMatMult": 0.04601097106933594,
|
||||
"WeightCoalescing": 0.049105167388916016,
|
||||
"ZeroSizeTensorElimination": 0.0003685951232910156
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 41972,
|
||||
"StaticProfiler::AifUb": 17.11549949645996,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 27.70398712158203,
|
||||
"StaticProfiler::AverageDmaLength": 2723.7119140625,
|
||||
"StaticProfiler::AverageFractalPeUtilization": 96.97119140625,
|
||||
"StaticProfiler::AveragePartitionUtilization": 88.53791809082031,
|
||||
"StaticProfiler::AveragePeUtilization": 81.30671691894531,
|
||||
"StaticProfiler::DDRTransferBytes": 918042224,
|
||||
"StaticProfiler::InternalTransferBytes": 174444896,
|
||||
"StaticProfiler::LoadExpanded": 238040,
|
||||
"StaticProfiler::LocalizationEfficiency": 161.8649139404297,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 167.2097930908203,
|
||||
"StaticProfiler::StoreExpanded": 18609,
|
||||
"StaticProfiler::TotalDMAExpanded": 256649,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 57730,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50535,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 179,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 30304,
|
||||
"TilingProfiler::NumPfTransposes": 348,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 198,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 7835,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 5217,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 508,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 2110,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 59,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2266,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"topk": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.003599405288696289,
|
||||
"DMALocalityOpt": 0.003013134002685547,
|
||||
"DMAProfiler": 0.0034313201904296875,
|
||||
"DataStreaming": 0.006362199783325195,
|
||||
"DoNothing": 0.00013327598571777344,
|
||||
"ExpandISAMacro": 0.003620147705078125,
|
||||
"FactorizeBlkDims": 0.010854482650756836,
|
||||
"InferPSumTensor": 0.010383129119873047,
|
||||
"InferSharedMemLoc": 0.0029261112213134766,
|
||||
"InsertCoreBarrier": 0.003323793411254883,
|
||||
"LateLegalizeInst": 0.007294178009033203,
|
||||
"LateNeuronInstComb": 0.008392095565795898,
|
||||
"LegalizeSundaAccess": 0.013858556747436523,
|
||||
"LegalizeType": 0.009220361709594727,
|
||||
"LowerBroadcast": 0.0032749176025390625,
|
||||
"LowerIntrinsics": 0.0036187171936035156,
|
||||
"LowerTranspose": 0.0032770633697509766,
|
||||
"NeuronInstComb": 0.008351325988769531,
|
||||
"NeuronLICM": 0.009110689163208008,
|
||||
"NeuronSimplifyPredicates": 0.0036361217498779297,
|
||||
"NeuronValueNumbering": 0.0038361549377441406,
|
||||
"SFKVectorizer": 0.031067371368408203,
|
||||
"SimpleAllReduceTiling": 0.0040929317474365234,
|
||||
"SimplifyNeuronTensor": 0.057866573333740234,
|
||||
"SpillPSum": 0.02007007598876953,
|
||||
"WeightCoalescing": 0.003391265869140625
|
||||
}
|
||||
}
|
||||
}
|
||||
3
token_generation_model/_tp0_bk0/graph.neff
Normal file
3
token_generation_model/_tp0_bk0/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:29d9d8ef7e5a0b735090051a93d177f1b4866fe07fc6988421cb47de73a1953f
|
||||
size 3185664
|
||||
4225
token_generation_model/_tp0_bk0/log-neuron-cc.txt
Normal file
4225
token_generation_model/_tp0_bk0/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
token_generation_model/_tp0_bk0/metaneff.pb
Normal file
3
token_generation_model/_tp0_bk0/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2751bdd6298e760b3ed04356f31f6ffbb7d7c37e3002f4e686a4fba8c24c2240
|
||||
size 2465286
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:166776f75e13c12167868e06c05edd9ecd84b1f5005be8ff59ffb090e4b60e00
|
||||
size 2443667
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:29d9d8ef7e5a0b735090051a93d177f1b4866fe07fc6988421cb47de73a1953f
|
||||
size 3185664
|
||||
224
token_generation_model/_tp0_bk0/neuron_config.json
Normal file
224
token_generation_model/_tp0_bk0/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 4,
|
||||
"bucket_n_active_tokens": false,
|
||||
"buckets": [
|
||||
128
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 1,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": null,
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": false,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 1,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": [
|
||||
128
|
||||
],
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
3
token_generation_model/_tp0_bk0/wrapped_neff.hlo
Normal file
3
token_generation_model/_tp0_bk0/wrapped_neff.hlo
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3d7d771432f2a5b315c36747141a382cd677d56bd52622fbc82564ea1e8d3890
|
||||
size 3380710
|
||||
1
token_generation_model/_tp0_bk1/command.txt
Normal file
1
token_generation_model/_tp0_bk1/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_f53407701fa4882a24c0+55c11e15.hlo_module.pb --output model.MODULE_f53407701fa4882a24c0+55c11e15.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk1/log-neuron-cc.txt"]
|
||||
590
token_generation_model/_tp0_bk1/global_metric_store.json
Normal file
590
token_generation_model/_tp0_bk1/global_metric_store.json
Normal file
@@ -0,0 +1,590 @@
|
||||
{
|
||||
"Average": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 97.62759399414063,
|
||||
"StaticProfiler::AveragePartitionUtilization": 90.06017303466797,
|
||||
"StaticProfiler::AveragePeUtilization": 82.14405822753906,
|
||||
"StaticProfiler::LocalizationEfficiency": 160.6605224609375,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 165.92486572265625,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
|
||||
}
|
||||
},
|
||||
"Count": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 1,
|
||||
"StaticProfiler::AveragePartitionUtilization": 1,
|
||||
"StaticProfiler::AveragePeUtilization": 1,
|
||||
"StaticProfiler::LocalizationEfficiency": 1,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
|
||||
}
|
||||
},
|
||||
"Sum": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 5.160665512084961,
|
||||
"AffinePredicateResolution": 0.7100803852081299,
|
||||
"AliasDependencyElimination": 0.0032160282135009766,
|
||||
"AliasDependencyInduction": 17.29242706298828,
|
||||
"AliasDependencyReset": 18.28659439086914,
|
||||
"BFComputeCutting": 0.24476861953735352,
|
||||
"BirCodeGenLoop": 6.947810173034668,
|
||||
"CCOpFusion": 0.6929187774658203,
|
||||
"CanonicalizeConv": 2.9000000722589903e-05,
|
||||
"CanonicalizeDAGForPGTiling": 0.21501421928405762,
|
||||
"CanonicalizeForTensorizer": 0.0008399999933317304,
|
||||
"CanonicalizeIR": 0.8826379776000977,
|
||||
"Canonicalizer": 0.014088000170886517,
|
||||
"CoalesceCCOp": 0.5657715797424316,
|
||||
"CommuteConcat": 0.028485536575317383,
|
||||
"DMALocalityOpt": 0.042168617248535156,
|
||||
"DMAProfiler": 0.1519615650177002,
|
||||
"DMATilingProfiler": 0.09857678413391113,
|
||||
"DataLocalityOpt": 2.827857255935669,
|
||||
"DataStreaming": 0.20627403259277344,
|
||||
"DeConcat": 0.08719110488891602,
|
||||
"DeadCodeElimination": 0.03139615058898926,
|
||||
"DeadStoreElimination": 1.1847929954528809,
|
||||
"DelinearIndices": 0.4130244255065918,
|
||||
"Delinearization": 0.35245656967163086,
|
||||
"DelinearizeSPMD": 0.4181056022644043,
|
||||
"DoNothing": 0.00046539306640625,
|
||||
"DramToDramTranspose": 0.30753135681152344,
|
||||
"DumpGraphAndMetadata": 0.2011098861694336,
|
||||
"EliminateDivs": 1.3804805278778076,
|
||||
"ExpandBatchNorm": 1.1923854351043701,
|
||||
"ExpandISAMacro": 0.08926725387573242,
|
||||
"FactorizeBlkDims": 0.5252444744110107,
|
||||
"FactorizeThreadAxesInFreeDims": 0.1739037036895752,
|
||||
"FlattenMacroLoop": 0.07839345932006836,
|
||||
"GenericAccessSimplifier": 0.025495052337646484,
|
||||
"HoistCompute": 6.199999916134402e-05,
|
||||
"IdentifyCrossPassTensors": 0.0005760000203736126,
|
||||
"InferInitValue": 1.3857665061950684,
|
||||
"InferIntrinsicOnCC": 0.646845817565918,
|
||||
"InferNeuronTensor": 2.0493955612182617,
|
||||
"InferNonlocalTensors": 6.630102634429932,
|
||||
"InferPSumTensor": 1.2878854274749756,
|
||||
"InferShardAxis": 10.588101387023926,
|
||||
"InferSharedMemLoc": 0.10591363906860352,
|
||||
"InlineNativeKernels": 0.0488896369934082,
|
||||
"InsertCoreBarrier": 0.3818776607513428,
|
||||
"InsertIOTransposes": 0.8546113967895508,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.36914730072021484,
|
||||
"InsertLocalTransposes": 1.5642907619476318,
|
||||
"InsertOffloadedTransposes": 0.12299966812133789,
|
||||
"LICM": 0.12427496910095215,
|
||||
"LateLegalizeInst": 0.4006483554840088,
|
||||
"LateLegalizePostSplit": 0.09814929962158203,
|
||||
"LateLowerReshapeOp": 0.0431976318359375,
|
||||
"LateLowerTensorOp": 5.883858680725098,
|
||||
"LateNeuronInstComb": 1.04233980178833,
|
||||
"LayoutPreprocessing": 1.2663531303405762,
|
||||
"LayoutPreprocessingAndAnalysis": 1.9893858432769775,
|
||||
"LayoutRequirementAnalysis": 0.6988728046417236,
|
||||
"LegalizeCCOpLayout": 1.499981164932251,
|
||||
"LegalizeOpLevelAlias": 0.6842360496520996,
|
||||
"LegalizePartitionReduce": 0.09090352058410645,
|
||||
"LegalizeSundaAccess": 3.133746862411499,
|
||||
"LegalizeSundaMacro": 0.6402661800384521,
|
||||
"LegalizeType": 0.1407451629638672,
|
||||
"LocalLayoutOpt": 0.7939743995666504,
|
||||
"LoopFusion": 0.29726386070251465,
|
||||
"LoopSplitting": 0.08333015441894531,
|
||||
"LowerBroadcast": 0.06013798713684082,
|
||||
"LowerCCOpBlockAxis": 1.4568629264831543,
|
||||
"LowerComplexBroadcast": 0.06818914413452148,
|
||||
"LowerIntrinsics": 0.9435675144195557,
|
||||
"LowerShardAxis": 0.22619009017944336,
|
||||
"LowerTensorOp": 3.0802407264709473,
|
||||
"LowerToSendRecv": 0.19478487968444824,
|
||||
"LowerTranspose": 0.5133018493652344,
|
||||
"MacroGeneration": 4.587047100067139,
|
||||
"MaskPropagation": 0.1672959327697754,
|
||||
"MemcastMotion": 0.00015999999595806003,
|
||||
"MemcpyElimination": 29.92277717590332,
|
||||
"MutateDataType": 0.04070854187011719,
|
||||
"NeuronAliasDependencyInduction": 0.018942594528198242,
|
||||
"NeuronAliasDependencyReset": 0.02425408363342285,
|
||||
"NeuronInstComb": 0.4308168888092041,
|
||||
"NeuronLICM": 0.29584765434265137,
|
||||
"NeuronLoopFusion": 1.117248296737671,
|
||||
"NeuronLoopInterchange": 0.0636141300201416,
|
||||
"NeuronSimplifier": 0.704033374786377,
|
||||
"NeuronSimplifyPredicates": 0.2319803237915039,
|
||||
"NeuronValueNumbering": 0.11443066596984863,
|
||||
"OptimizeAliasedCopyChain": 0.2889397144317627,
|
||||
"OptimizeNKIKernels": 1.2374975681304932,
|
||||
"PAGLayoutOpt": 18.6531982421875,
|
||||
"PComputeCutting": 0.5685915946960449,
|
||||
"PGLayoutTilingPipeline": 53.59619903564453,
|
||||
"PGTiling": 11.116703033447266,
|
||||
"PadElimination": 0.012248039245605469,
|
||||
"ParAxesAnnotation": 17.06294822692871,
|
||||
"PartialLoopFusion": 1.3277881145477295,
|
||||
"PartialSimdFusion": 0.7600483894348145,
|
||||
"PenguinizeFunctions": 0.001062000053934753,
|
||||
"PerfectLoopNest": 0.061771392822265625,
|
||||
"PruneFunctions": 0.0004830000107176602,
|
||||
"RecognizeOpIdiom": 0.12406373023986816,
|
||||
"Recompute": 0.00886678695678711,
|
||||
"RelaxPredicates": 0.11265873908996582,
|
||||
"Rematerialization": 0.18548583984375,
|
||||
"RemoveOptimizationBarriers": 0.009134000167250633,
|
||||
"RemoveShardedPartitionAxes": 1.4166390895843506,
|
||||
"ReshapeWeights": 0.02198624610900879,
|
||||
"ResolveAccessConflict": 0.20142865180969238,
|
||||
"ResolveComplicatePredicates": 0.6187732219696045,
|
||||
"RewriteReplicationMatmul": 0.04161477088928223,
|
||||
"RewriteWeights": 0.0676727294921875,
|
||||
"SFKVectorizer": 11.220393180847168,
|
||||
"ScatterMotion": 0.003527000080794096,
|
||||
"ShardingPropagationAnalysis": 0.8252537250518799,
|
||||
"SimpleAllReduceTiling": 0.1859297752380371,
|
||||
"Simplifier": 0.09509730339050293,
|
||||
"SimplifyMacroPredicates": 0.26697540283203125,
|
||||
"SimplifyNeuronTensor": 0.41252875328063965,
|
||||
"SimplifySlice": 0.02629375457763672,
|
||||
"SimplifyTensor": 0.42005443572998047,
|
||||
"SpillPSum": 0.5884366035461426,
|
||||
"SplitAPUnionSets": 0.4571385383605957,
|
||||
"SplitAccGrp": 0.052629709243774414,
|
||||
"StaticProfiler": 0.12954020500183105,
|
||||
"StaticTransposeLocalTensor": 0.36740827560424805,
|
||||
"SundaISel": 1.4737660884857178,
|
||||
"TCTransform": 0.031054019927978516,
|
||||
"TensorInitialization": 0.18288874626159668,
|
||||
"TensorOpSimplifier": 3.2880165576934814,
|
||||
"TensorOpTransform": 18.079126358032227,
|
||||
"TensorizerLegalizationPass": 0.0004990000161342323,
|
||||
"TileCCOps": 0.17596793174743652,
|
||||
"TilingProfiler": 0.3990769386291504,
|
||||
"TransformConvOp": 1.2077322006225586,
|
||||
"TritiumFusion": 0.2976958751678467,
|
||||
"ValueNumbering": 0.10138130187988281,
|
||||
"VectorizeDMA": 0.8864037990570068,
|
||||
"VectorizeMatMult": 0.05278921127319336,
|
||||
"VerifySupportedOps": 0.0004729999927803874,
|
||||
"WeightCoalescing": 0.08338785171508789,
|
||||
"ZeroSizeTensorElimination": 0.0006525516510009766,
|
||||
"algsimp": 0.0014349999837577343,
|
||||
"batchnorm_expander": 0.000818000000435859,
|
||||
"boundary-marker-removal": 0.0003020000003743917,
|
||||
"call-inliner": 0.00018099999579135329,
|
||||
"canonicalize-boundary-marker": 0.0005639999872073531,
|
||||
"collective-stream-id-checker": 0.002294000005349517,
|
||||
"comparison-expander": 0.0012120000319555402,
|
||||
"computation-deduplicator": 0.0011109999613836408,
|
||||
"config-lowering": 0.00042299999040551484,
|
||||
"constant_folding": 0.0001289999927394092,
|
||||
"cse": 0.0007570000016130507,
|
||||
"dce": 7.300000288523734e-05,
|
||||
"dynamic-slice-transpose": 0.0002640000020619482,
|
||||
"eliminate-redundant-compare": 0.0001250000059371814,
|
||||
"emit-offloaded-dropout": 0.0006559999892488122,
|
||||
"flatten-call-graph": 0.0003330000035930425,
|
||||
"fuse-send-recv": 0.013015000149607658,
|
||||
"hilo-conditional-to-select": 0.00016799999866634607,
|
||||
"hilo::LegalizeAlias": 0.0036430000327527523,
|
||||
"hilo::NeuronInstCombine": 1.1000000085914508e-05,
|
||||
"hilo::NeuronOpFusion": 0.00011600000289035961,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.0003260000084992498,
|
||||
"hilo::ScheduleFusion": 3.300000025774352e-05,
|
||||
"hilo::SixtyFourHack": 0.0013909999979659915,
|
||||
"hilo::VerifyAliasing": 0.00014600000577047467,
|
||||
"hlo-mac-count": 0.019317999482154846,
|
||||
"io-con-pipe-begin": 0.0002589999930933118,
|
||||
"io-con-pipe-end": 9.999999974752427e-07,
|
||||
"io-layout-normalization": 0.07799100130796432,
|
||||
"legalize-ccops-for-tensorizer": 5.2999999752501026e-05,
|
||||
"legalize-compare": 0.0006939999875612557,
|
||||
"lower-argminmax-custom-call": 0.0002629999944474548,
|
||||
"map-inline": 0.06807900220155716,
|
||||
"metadata-naming": 0.07221200317144394,
|
||||
"mlir::detail::OpToOpPassAdaptor": 0.007284999825060368,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.6235949993133545,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.00916799996048212,
|
||||
"mlir::mhlo::LowerComplexPass": 0.0006799999973736703,
|
||||
"native-to-custom-softmax": 0.0017719999887049198,
|
||||
"native-to-custom-softmax-dx": 0.0017500000540167093,
|
||||
"neuron-hlo-verifier": 0.2390509992837906,
|
||||
"operand_upcaster": 0.070872001349926,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 9.999999747378752e-06,
|
||||
"post-partition-simplification": 0.21146699786186218,
|
||||
"pre-hlo-begin": 8.600000001024455e-05,
|
||||
"pre-hlo-end": 9.999999974752427e-07,
|
||||
"replace-minimum-constant": 0.00019700000120792538,
|
||||
"reshape-mover": 5.400000009103678e-05,
|
||||
"simplify-concat": 0.002297000028192997,
|
||||
"simplify-while-loops": 4.600000102072954e-05,
|
||||
"transform-variadic-reduce": 0.00037900000461377203,
|
||||
"tuple-simplifier": 0.0001340000017080456,
|
||||
"unpack-nested-aws-ntwsr": 0.00025400001322850585,
|
||||
"unroll-while-loop": 7.000000096013537e-06
|
||||
},
|
||||
"hilo": {
|
||||
"HloMacCount": 1766645760.0,
|
||||
"Traffic": 1252618368.0
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 42842,
|
||||
"StaticProfiler::AifUb": 17.765029907226563,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 28.541391372680664,
|
||||
"StaticProfiler::AverageDmaLength": 3512.794677734375,
|
||||
"StaticProfiler::DDRTransferBytes": 924925552,
|
||||
"StaticProfiler::InternalTransferBytes": 182108768,
|
||||
"StaticProfiler::LoadExpanded": 181148,
|
||||
"StaticProfiler::StoreExpanded": 4273,
|
||||
"StaticProfiler::TotalDMAExpanded": 185421,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 57693,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50217,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 123,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 30976,
|
||||
"TilingProfiler::NumPfTransposes": 348,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 198,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 7892,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 5666,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 564,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 115,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2295,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"all": {
|
||||
"compiletime": {
|
||||
"CanonicalizeConv": 2.9000000722589903e-05,
|
||||
"CanonicalizeForTensorizer": 0.0008399999933317304,
|
||||
"Canonicalizer": 0.014088000170886517,
|
||||
"HoistCompute": 6.199999916134402e-05,
|
||||
"IdentifyCrossPassTensors": 0.0005760000203736126,
|
||||
"MemcastMotion": 0.00015999999595806003,
|
||||
"PenguinizeFunctions": 0.001062000053934753,
|
||||
"PruneFunctions": 0.0004830000107176602,
|
||||
"RemoveOptimizationBarriers": 0.009134000167250633,
|
||||
"ScatterMotion": 0.003527000080794096,
|
||||
"TensorizerLegalizationPass": 0.0004990000161342323,
|
||||
"VerifySupportedOps": 0.0004729999927803874,
|
||||
"algsimp": 0.0014349999837577343,
|
||||
"batchnorm_expander": 0.000818000000435859,
|
||||
"boundary-marker-removal": 0.0003020000003743917,
|
||||
"call-inliner": 0.00018099999579135329,
|
||||
"canonicalize-boundary-marker": 0.0005639999872073531,
|
||||
"collective-stream-id-checker": 0.002294000005349517,
|
||||
"comparison-expander": 0.0012120000319555402,
|
||||
"computation-deduplicator": 0.0011109999613836408,
|
||||
"config-lowering": 0.00042299999040551484,
|
||||
"constant_folding": 0.0001289999927394092,
|
||||
"cse": 0.0007570000016130507,
|
||||
"dce": 7.300000288523734e-05,
|
||||
"dynamic-slice-transpose": 0.0002640000020619482,
|
||||
"eliminate-redundant-compare": 0.0001250000059371814,
|
||||
"emit-offloaded-dropout": 0.0006559999892488122,
|
||||
"flatten-call-graph": 0.0003330000035930425,
|
||||
"fuse-send-recv": 0.013015000149607658,
|
||||
"hilo-conditional-to-select": 0.00016799999866634607,
|
||||
"hilo::LegalizeAlias": 0.0036430000327527523,
|
||||
"hilo::NeuronInstCombine": 1.1000000085914508e-05,
|
||||
"hilo::NeuronOpFusion": 0.00011600000289035961,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.0003260000084992498,
|
||||
"hilo::ScheduleFusion": 3.300000025774352e-05,
|
||||
"hilo::SixtyFourHack": 0.0013909999979659915,
|
||||
"hilo::VerifyAliasing": 0.00014600000577047467,
|
||||
"hlo-mac-count": 0.019317999482154846,
|
||||
"io-con-pipe-begin": 0.0002589999930933118,
|
||||
"io-con-pipe-end": 9.999999974752427e-07,
|
||||
"io-layout-normalization": 0.07799100130796432,
|
||||
"legalize-ccops-for-tensorizer": 5.2999999752501026e-05,
|
||||
"legalize-compare": 0.0006939999875612557,
|
||||
"lower-argminmax-custom-call": 0.0002629999944474548,
|
||||
"map-inline": 0.06807900220155716,
|
||||
"metadata-naming": 0.07221200317144394,
|
||||
"mlir::detail::OpToOpPassAdaptor": 0.007284999825060368,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.6235949993133545,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.00916799996048212,
|
||||
"mlir::mhlo::LowerComplexPass": 0.0006799999973736703,
|
||||
"native-to-custom-softmax": 0.0017719999887049198,
|
||||
"native-to-custom-softmax-dx": 0.0017500000540167093,
|
||||
"neuron-hlo-verifier": 0.2390509992837906,
|
||||
"operand_upcaster": 0.070872001349926,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 9.999999747378752e-06,
|
||||
"post-partition-simplification": 0.21146699786186218,
|
||||
"pre-hlo-begin": 8.600000001024455e-05,
|
||||
"pre-hlo-end": 9.999999974752427e-07,
|
||||
"replace-minimum-constant": 0.00019700000120792538,
|
||||
"reshape-mover": 5.400000009103678e-05,
|
||||
"simplify-concat": 0.002297000028192997,
|
||||
"simplify-while-loops": 4.600000102072954e-05,
|
||||
"transform-variadic-reduce": 0.00037900000461377203,
|
||||
"tuple-simplifier": 0.0001340000017080456,
|
||||
"unpack-nested-aws-ntwsr": 0.00025400001322850585,
|
||||
"unroll-while-loop": 7.000000096013537e-06
|
||||
}
|
||||
},
|
||||
"cumsum": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.0002148151397705078,
|
||||
"DMALocalityOpt": 0.0001659393310546875,
|
||||
"DMAProfiler": 0.0008566379547119141,
|
||||
"DataStreaming": 0.00026869773864746094,
|
||||
"DoNothing": 0.00015687942504882813,
|
||||
"ExpandISAMacro": 0.0004825592041015625,
|
||||
"FactorizeBlkDims": 0.0004360675811767578,
|
||||
"InferPSumTensor": 0.0005650520324707031,
|
||||
"InferSharedMemLoc": 0.0002880096435546875,
|
||||
"InsertCoreBarrier": 0.0002799034118652344,
|
||||
"LateLegalizeInst": 0.0003802776336669922,
|
||||
"LateNeuronInstComb": 0.0006039142608642578,
|
||||
"LegalizeSundaAccess": 0.001421213150024414,
|
||||
"LegalizeType": 0.0002644062042236328,
|
||||
"LowerBroadcast": 0.0002384185791015625,
|
||||
"LowerIntrinsics": 0.0002446174621582031,
|
||||
"LowerTranspose": 0.00022077560424804688,
|
||||
"NeuronInstComb": 0.0005753040313720703,
|
||||
"NeuronLICM": 0.00037217140197753906,
|
||||
"NeuronSimplifyPredicates": 0.0022978782653808594,
|
||||
"NeuronValueNumbering": 0.0004074573516845703,
|
||||
"SFKVectorizer": 0.002466440200805664,
|
||||
"SimpleAllReduceTiling": 0.0002262592315673828,
|
||||
"SimplifyNeuronTensor": 0.0005323886871337891,
|
||||
"SpillPSum": 0.00045490264892578125,
|
||||
"WeightCoalescing": 0.0002465248107910156
|
||||
}
|
||||
},
|
||||
"sg00": {
|
||||
"hilo": {
|
||||
"ArithmeticIntensity": 2.8207247257232666,
|
||||
"HloMacCount": 1766645760.0,
|
||||
"Traffic": 1252618368.0
|
||||
}
|
||||
},
|
||||
"sg0000": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 5.160665512084961,
|
||||
"AffinePredicateResolution": 0.7100803852081299,
|
||||
"AliasDependencyElimination": 0.0032160282135009766,
|
||||
"AliasDependencyInduction": 17.29242706298828,
|
||||
"AliasDependencyReset": 18.28659439086914,
|
||||
"BFComputeCutting": 0.24476861953735352,
|
||||
"BirCodeGenLoop": 6.947810173034668,
|
||||
"CCOpFusion": 0.6929187774658203,
|
||||
"CanonicalizeDAGForPGTiling": 0.21501421928405762,
|
||||
"CanonicalizeIR": 0.8826379776000977,
|
||||
"CoalesceCCOp": 0.5618836879730225,
|
||||
"CommuteConcat": 0.028485536575317383,
|
||||
"DMALocalityOpt": 0.03933858871459961,
|
||||
"DMAProfiler": 0.14742827415466309,
|
||||
"DMATilingProfiler": 0.09857678413391113,
|
||||
"DataLocalityOpt": 2.827857255935669,
|
||||
"DataStreaming": 0.19969892501831055,
|
||||
"DeConcat": 0.08719110488891602,
|
||||
"DeadCodeElimination": 0.03139615058898926,
|
||||
"DeadStoreElimination": 1.1847929954528809,
|
||||
"DelinearIndices": 0.4130244255065918,
|
||||
"Delinearization": 0.35245656967163086,
|
||||
"DelinearizeSPMD": 0.4181056022644043,
|
||||
"DoNothing": 0.0001385211944580078,
|
||||
"DramToDramTranspose": 0.30753135681152344,
|
||||
"DumpGraphAndMetadata": 0.2011098861694336,
|
||||
"EliminateDivs": 1.3804805278778076,
|
||||
"ExpandBatchNorm": 1.1923854351043701,
|
||||
"ExpandISAMacro": 0.08506417274475098,
|
||||
"FactorizeBlkDims": 0.5093107223510742,
|
||||
"FactorizeThreadAxesInFreeDims": 0.1739037036895752,
|
||||
"FlattenMacroLoop": 0.07839345932006836,
|
||||
"GenericAccessSimplifier": 0.025495052337646484,
|
||||
"InferInitValue": 1.3857665061950684,
|
||||
"InferIntrinsicOnCC": 0.646845817565918,
|
||||
"InferNeuronTensor": 2.0493955612182617,
|
||||
"InferNonlocalTensors": 6.630102634429932,
|
||||
"InferPSumTensor": 1.275925636291504,
|
||||
"InferShardAxis": 10.588101387023926,
|
||||
"InferSharedMemLoc": 0.102783203125,
|
||||
"InlineNativeKernels": 0.0488896369934082,
|
||||
"InsertCoreBarrier": 0.3783996105194092,
|
||||
"InsertIOTransposes": 0.8546113967895508,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.36914730072021484,
|
||||
"InsertLocalTransposes": 1.5642907619476318,
|
||||
"InsertOffloadedTransposes": 0.12299966812133789,
|
||||
"LICM": 0.12427496910095215,
|
||||
"LateLegalizeInst": 0.3925168514251709,
|
||||
"LateLegalizePostSplit": 0.09814929962158203,
|
||||
"LateLowerReshapeOp": 0.0431976318359375,
|
||||
"LateLowerTensorOp": 5.883858680725098,
|
||||
"LateNeuronInstComb": 1.0332238674163818,
|
||||
"LayoutPreprocessing": 1.2663531303405762,
|
||||
"LayoutPreprocessingAndAnalysis": 1.9893858432769775,
|
||||
"LayoutRequirementAnalysis": 0.6988728046417236,
|
||||
"LegalizeCCOpLayout": 1.499981164932251,
|
||||
"LegalizeOpLevelAlias": 0.6842360496520996,
|
||||
"LegalizePartitionReduce": 0.09090352058410645,
|
||||
"LegalizeSundaAccess": 3.1170387268066406,
|
||||
"LegalizeSundaMacro": 0.6402661800384521,
|
||||
"LegalizeType": 0.1309196949005127,
|
||||
"LocalLayoutOpt": 0.7939743995666504,
|
||||
"LoopFusion": 0.29726386070251465,
|
||||
"LoopSplitting": 0.08333015441894531,
|
||||
"LowerBroadcast": 0.05664563179016113,
|
||||
"LowerCCOpBlockAxis": 1.4568629264831543,
|
||||
"LowerComplexBroadcast": 0.06818914413452148,
|
||||
"LowerIntrinsics": 0.9396772384643555,
|
||||
"LowerShardAxis": 0.22619009017944336,
|
||||
"LowerTensorOp": 3.0802407264709473,
|
||||
"LowerToSendRecv": 0.19478487968444824,
|
||||
"LowerTranspose": 0.5097734928131104,
|
||||
"MacroGeneration": 4.587047100067139,
|
||||
"MaskPropagation": 0.1672959327697754,
|
||||
"MemcpyElimination": 29.92277717590332,
|
||||
"MutateDataType": 0.04070854187011719,
|
||||
"NeuronAliasDependencyInduction": 0.018942594528198242,
|
||||
"NeuronAliasDependencyReset": 0.02425408363342285,
|
||||
"NeuronInstComb": 0.42151784896850586,
|
||||
"NeuronLICM": 0.2860753536224365,
|
||||
"NeuronLoopFusion": 1.117248296737671,
|
||||
"NeuronLoopInterchange": 0.0636141300201416,
|
||||
"NeuronSimplifier": 0.704033374786377,
|
||||
"NeuronSimplifyPredicates": 0.22606325149536133,
|
||||
"NeuronValueNumbering": 0.11009359359741211,
|
||||
"OptimizeAliasedCopyChain": 0.2889397144317627,
|
||||
"OptimizeNKIKernels": 1.2374975681304932,
|
||||
"PAGLayoutOpt": 18.6531982421875,
|
||||
"PComputeCutting": 0.5685915946960449,
|
||||
"PGLayoutTilingPipeline": 53.59619903564453,
|
||||
"PGTiling": 11.116703033447266,
|
||||
"PadElimination": 0.012248039245605469,
|
||||
"ParAxesAnnotation": 17.06294822692871,
|
||||
"PartialLoopFusion": 1.3277881145477295,
|
||||
"PartialSimdFusion": 0.7600483894348145,
|
||||
"PerfectLoopNest": 0.061771392822265625,
|
||||
"RecognizeOpIdiom": 0.12406373023986816,
|
||||
"Recompute": 0.00886678695678711,
|
||||
"RelaxPredicates": 0.11265873908996582,
|
||||
"Rematerialization": 0.18548583984375,
|
||||
"RemoveShardedPartitionAxes": 1.4166390895843506,
|
||||
"ReshapeWeights": 0.02198624610900879,
|
||||
"ResolveAccessConflict": 0.20142865180969238,
|
||||
"ResolveComplicatePredicates": 0.6187732219696045,
|
||||
"RewriteReplicationMatmul": 0.04161477088928223,
|
||||
"RewriteWeights": 0.0676727294921875,
|
||||
"SFKVectorizer": 11.18409252166748,
|
||||
"ShardingPropagationAnalysis": 0.8252537250518799,
|
||||
"SimpleAllReduceTiling": 0.18207907676696777,
|
||||
"Simplifier": 0.09509730339050293,
|
||||
"SimplifyMacroPredicates": 0.26697540283203125,
|
||||
"SimplifyNeuronTensor": 0.3594348430633545,
|
||||
"SimplifySlice": 0.02629375457763672,
|
||||
"SimplifyTensor": 0.42005443572998047,
|
||||
"SpillPSum": 0.5635168552398682,
|
||||
"SplitAPUnionSets": 0.4571385383605957,
|
||||
"SplitAccGrp": 0.052629709243774414,
|
||||
"StaticProfiler": 0.12954020500183105,
|
||||
"StaticTransposeLocalTensor": 0.36740827560424805,
|
||||
"SundaISel": 1.4737660884857178,
|
||||
"TCTransform": 0.031054019927978516,
|
||||
"TensorInitialization": 0.18288874626159668,
|
||||
"TensorOpSimplifier": 3.2880165576934814,
|
||||
"TensorOpTransform": 18.079126358032227,
|
||||
"TileCCOps": 0.17596793174743652,
|
||||
"TilingProfiler": 0.3990769386291504,
|
||||
"TransformConvOp": 1.2077322006225586,
|
||||
"TritiumFusion": 0.2976958751678467,
|
||||
"ValueNumbering": 0.10138130187988281,
|
||||
"VectorizeDMA": 0.8864037990570068,
|
||||
"VectorizeMatMult": 0.05278921127319336,
|
||||
"WeightCoalescing": 0.07968640327453613,
|
||||
"ZeroSizeTensorElimination": 0.0006525516510009766
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 42842,
|
||||
"StaticProfiler::AifUb": 17.765029907226563,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 28.541391372680664,
|
||||
"StaticProfiler::AverageDmaLength": 3512.794677734375,
|
||||
"StaticProfiler::AverageFractalPeUtilization": 97.62759399414063,
|
||||
"StaticProfiler::AveragePartitionUtilization": 90.06017303466797,
|
||||
"StaticProfiler::AveragePeUtilization": 82.14405822753906,
|
||||
"StaticProfiler::DDRTransferBytes": 924925552,
|
||||
"StaticProfiler::InternalTransferBytes": 182108768,
|
||||
"StaticProfiler::LoadExpanded": 181148,
|
||||
"StaticProfiler::LocalizationEfficiency": 160.6605224609375,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 165.92486572265625,
|
||||
"StaticProfiler::StoreExpanded": 4273,
|
||||
"StaticProfiler::TotalDMAExpanded": 185421,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 57693,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50217,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 123,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 30976,
|
||||
"TilingProfiler::NumPfTransposes": 348,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 198,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 7892,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 5666,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 564,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 115,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2295,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"topk": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.003673076629638672,
|
||||
"DMALocalityOpt": 0.0026640892028808594,
|
||||
"DMAProfiler": 0.0036766529083251953,
|
||||
"DataStreaming": 0.00630640983581543,
|
||||
"DoNothing": 0.00016999244689941406,
|
||||
"ExpandISAMacro": 0.003720521926879883,
|
||||
"FactorizeBlkDims": 0.015497684478759766,
|
||||
"InferPSumTensor": 0.011394739151000977,
|
||||
"InferSharedMemLoc": 0.002842426300048828,
|
||||
"InsertCoreBarrier": 0.0031981468200683594,
|
||||
"LateLegalizeInst": 0.0077512264251708984,
|
||||
"LateNeuronInstComb": 0.008512020111083984,
|
||||
"LegalizeSundaAccess": 0.015286922454833984,
|
||||
"LegalizeType": 0.00956106185913086,
|
||||
"LowerBroadcast": 0.003253936767578125,
|
||||
"LowerIntrinsics": 0.003645658493041992,
|
||||
"LowerTranspose": 0.0033075809478759766,
|
||||
"NeuronInstComb": 0.008723735809326172,
|
||||
"NeuronLICM": 0.009400129318237305,
|
||||
"NeuronSimplifyPredicates": 0.0036191940307617188,
|
||||
"NeuronValueNumbering": 0.003929615020751953,
|
||||
"SFKVectorizer": 0.033834218978881836,
|
||||
"SimpleAllReduceTiling": 0.003624439239501953,
|
||||
"SimplifyNeuronTensor": 0.05256152153015137,
|
||||
"SpillPSum": 0.024464845657348633,
|
||||
"WeightCoalescing": 0.003454923629760742
|
||||
}
|
||||
}
|
||||
}
|
||||
3
token_generation_model/_tp0_bk1/graph.neff
Normal file
3
token_generation_model/_tp0_bk1/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cda6f9bc52718ce93e1fb58a7aa253d606cf545bda03b396e531716e6a9ce678
|
||||
size 3154944
|
||||
4469
token_generation_model/_tp0_bk1/log-neuron-cc.txt
Normal file
4469
token_generation_model/_tp0_bk1/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
token_generation_model/_tp0_bk1/metaneff.pb
Normal file
3
token_generation_model/_tp0_bk1/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:fd8b7f01387ce1b80997167a938d7a05e712ade8a8271b5da3b01c446894cbd8
|
||||
size 2464555
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:d9364e7ab828d2cbc2c983e088e9c293cdaddc1137d948ee2d3c625e9eab02f7
|
||||
size 2550451
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cda6f9bc52718ce93e1fb58a7aa253d606cf545bda03b396e531716e6a9ce678
|
||||
size 3154944
|
||||
224
token_generation_model/_tp0_bk1/neuron_config.json
Normal file
224
token_generation_model/_tp0_bk1/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 4,
|
||||
"bucket_n_active_tokens": false,
|
||||
"buckets": [
|
||||
256
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 1,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": null,
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": false,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 1,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": [
|
||||
256
|
||||
],
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
token_generation_model/_tp0_bk2/command.txt
Normal file
1
token_generation_model/_tp0_bk2/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_c60e20a111715e620aa2+6bb8acc9.hlo_module.pb --output model.MODULE_c60e20a111715e620aa2+6bb8acc9.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk2/log-neuron-cc.txt"]
|
||||
590
token_generation_model/_tp0_bk2/global_metric_store.json
Normal file
590
token_generation_model/_tp0_bk2/global_metric_store.json
Normal file
@@ -0,0 +1,590 @@
|
||||
{
|
||||
"Average": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 97.78141784667969,
|
||||
"StaticProfiler::AveragePartitionUtilization": 90.33787536621094,
|
||||
"StaticProfiler::AveragePeUtilization": 81.002685546875,
|
||||
"StaticProfiler::LocalizationEfficiency": 155.71730041503906,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 160.65768432617188,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
|
||||
}
|
||||
},
|
||||
"Count": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 1,
|
||||
"StaticProfiler::AveragePartitionUtilization": 1,
|
||||
"StaticProfiler::AveragePeUtilization": 1,
|
||||
"StaticProfiler::LocalizationEfficiency": 1,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
|
||||
}
|
||||
},
|
||||
"Sum": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 6.775559425354004,
|
||||
"AffinePredicateResolution": 0.7124242782592773,
|
||||
"AliasDependencyElimination": 0.0037508010864257813,
|
||||
"AliasDependencyInduction": 17.88080596923828,
|
||||
"AliasDependencyReset": 19.01345443725586,
|
||||
"BFComputeCutting": 0.14321279525756836,
|
||||
"BirCodeGenLoop": 4.049878120422363,
|
||||
"CCOpFusion": 0.9675219058990479,
|
||||
"CanonicalizeConv": 0.0,
|
||||
"CanonicalizeDAGForPGTiling": 0.18202924728393555,
|
||||
"CanonicalizeForTensorizer": 0.0005849999724887311,
|
||||
"CanonicalizeIR": 0.8085732460021973,
|
||||
"Canonicalizer": 0.015991000458598137,
|
||||
"CoalesceCCOp": 0.25977158546447754,
|
||||
"CommuteConcat": 0.027615070343017578,
|
||||
"DMALocalityOpt": 0.048050880432128906,
|
||||
"DMAProfiler": 0.09339690208435059,
|
||||
"DMATilingProfiler": 0.09855914115905762,
|
||||
"DataLocalityOpt": 2.8309335708618164,
|
||||
"DataStreaming": 0.2996983528137207,
|
||||
"DeConcat": 0.08213448524475098,
|
||||
"DeadCodeElimination": 0.030428647994995117,
|
||||
"DeadStoreElimination": 1.051145315170288,
|
||||
"DelinearIndices": 0.5529322624206543,
|
||||
"Delinearization": 0.14493751525878906,
|
||||
"DelinearizeSPMD": 0.181105375289917,
|
||||
"DoNothing": 0.0003535747528076172,
|
||||
"DramToDramTranspose": 0.5176758766174316,
|
||||
"DumpGraphAndMetadata": 0.1749565601348877,
|
||||
"EliminateDivs": 1.6087908744812012,
|
||||
"ExpandBatchNorm": 1.096595048904419,
|
||||
"ExpandISAMacro": 0.09282922744750977,
|
||||
"FactorizeBlkDims": 0.5121583938598633,
|
||||
"FactorizeThreadAxesInFreeDims": 0.08890652656555176,
|
||||
"FlattenMacroLoop": 0.08152222633361816,
|
||||
"GenericAccessSimplifier": 0.02545022964477539,
|
||||
"HoistCompute": 9.899999713525176e-05,
|
||||
"IdentifyCrossPassTensors": 0.00011500000255182385,
|
||||
"InferInitValue": 1.2978432178497314,
|
||||
"InferIntrinsicOnCC": 0.7616860866546631,
|
||||
"InferNeuronTensor": 2.6053829193115234,
|
||||
"InferNonlocalTensors": 7.113053321838379,
|
||||
"InferPSumTensor": 4.450908660888672,
|
||||
"InferShardAxis": 12.849608421325684,
|
||||
"InferSharedMemLoc": 0.19225406646728516,
|
||||
"InlineNativeKernels": 0.08442234992980957,
|
||||
"InsertCoreBarrier": 0.15249347686767578,
|
||||
"InsertIOTransposes": 0.8695223331451416,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.49620485305786133,
|
||||
"InsertLocalTransposes": 1.934779167175293,
|
||||
"InsertOffloadedTransposes": 0.23423099517822266,
|
||||
"LICM": 0.1213834285736084,
|
||||
"LateLegalizeInst": 0.28924560546875,
|
||||
"LateLegalizePostSplit": 0.22411513328552246,
|
||||
"LateLowerReshapeOp": 0.033789873123168945,
|
||||
"LateLowerTensorOp": 4.297260761260986,
|
||||
"LateNeuronInstComb": 1.0230679512023926,
|
||||
"LayoutPreprocessing": 1.8572075366973877,
|
||||
"LayoutPreprocessingAndAnalysis": 2.2904410362243652,
|
||||
"LayoutRequirementAnalysis": 0.4237644672393799,
|
||||
"LegalizeCCOpLayout": 1.1935856342315674,
|
||||
"LegalizeOpLevelAlias": 0.7793729305267334,
|
||||
"LegalizePartitionReduce": 0.09432244300842285,
|
||||
"LegalizeSundaAccess": 1.0004379749298096,
|
||||
"LegalizeSundaMacro": 0.7198967933654785,
|
||||
"LegalizeType": 0.1766369342803955,
|
||||
"LocalLayoutOpt": 0.6666848659515381,
|
||||
"LoopFusion": 0.29988932609558105,
|
||||
"LoopSplitting": 0.05208778381347656,
|
||||
"LowerBroadcast": 0.08215570449829102,
|
||||
"LowerCCOpBlockAxis": 0.2862672805786133,
|
||||
"LowerComplexBroadcast": 0.09356045722961426,
|
||||
"LowerIntrinsics": 1.3606979846954346,
|
||||
"LowerShardAxis": 0.39397311210632324,
|
||||
"LowerTensorOp": 4.018380165100098,
|
||||
"LowerToSendRecv": 0.27350616455078125,
|
||||
"LowerTranspose": 0.5171041488647461,
|
||||
"MacroGeneration": 2.0922586917877197,
|
||||
"MaskPropagation": 0.12857651710510254,
|
||||
"MemcastMotion": 0.0002789999998640269,
|
||||
"MemcpyElimination": 30.280338287353516,
|
||||
"MutateDataType": 0.03657245635986328,
|
||||
"NeuronAliasDependencyInduction": 0.026292085647583008,
|
||||
"NeuronAliasDependencyReset": 0.03253936767578125,
|
||||
"NeuronInstComb": 0.36768054962158203,
|
||||
"NeuronLICM": 0.3019218444824219,
|
||||
"NeuronLoopFusion": 1.711080551147461,
|
||||
"NeuronLoopInterchange": 0.08134770393371582,
|
||||
"NeuronSimplifier": 0.5104148387908936,
|
||||
"NeuronSimplifyPredicates": 0.24121522903442383,
|
||||
"NeuronValueNumbering": 0.11496901512145996,
|
||||
"OptimizeAliasedCopyChain": 0.40732264518737793,
|
||||
"OptimizeNKIKernels": 1.0493371486663818,
|
||||
"PAGLayoutOpt": 18.23680305480957,
|
||||
"PComputeCutting": 0.5901892185211182,
|
||||
"PGLayoutTilingPipeline": 54.30891036987305,
|
||||
"PGTiling": 10.472368240356445,
|
||||
"PadElimination": 0.0187380313873291,
|
||||
"ParAxesAnnotation": 16.293506622314453,
|
||||
"PartialLoopFusion": 1.5408625602722168,
|
||||
"PartialSimdFusion": 0.8349573612213135,
|
||||
"PenguinizeFunctions": 0.0005789999850094318,
|
||||
"PerfectLoopNest": 0.07371997833251953,
|
||||
"PruneFunctions": 0.00018899999849963933,
|
||||
"RecognizeOpIdiom": 0.12681221961975098,
|
||||
"Recompute": 0.007548809051513672,
|
||||
"RelaxPredicates": 0.11768078804016113,
|
||||
"Rematerialization": 0.15080475807189941,
|
||||
"RemoveOptimizationBarriers": 0.0004290000069886446,
|
||||
"RemoveShardedPartitionAxes": 1.1734652519226074,
|
||||
"ReshapeWeights": 0.022508859634399414,
|
||||
"ResolveAccessConflict": 0.20145153999328613,
|
||||
"ResolveComplicatePredicates": 0.3191838264465332,
|
||||
"RewriteReplicationMatmul": 0.039293527603149414,
|
||||
"RewriteWeights": 0.0773766040802002,
|
||||
"SFKVectorizer": 11.08578872680664,
|
||||
"ScatterMotion": 0.005127000156790018,
|
||||
"ShardingPropagationAnalysis": 0.665086030960083,
|
||||
"SimpleAllReduceTiling": 0.07452917098999023,
|
||||
"Simplifier": 0.09160900115966797,
|
||||
"SimplifyMacroPredicates": 0.28479433059692383,
|
||||
"SimplifyNeuronTensor": 0.5389096736907959,
|
||||
"SimplifySlice": 0.02684783935546875,
|
||||
"SimplifyTensor": 0.2644212245941162,
|
||||
"SpillPSum": 1.0592920780181885,
|
||||
"SplitAPUnionSets": 1.0491511821746826,
|
||||
"SplitAccGrp": 0.055588722229003906,
|
||||
"StaticProfiler": 0.39172840118408203,
|
||||
"StaticTransposeLocalTensor": 0.7211995124816895,
|
||||
"SundaISel": 1.5357580184936523,
|
||||
"TCTransform": 0.029030561447143555,
|
||||
"TensorInitialization": 0.20336008071899414,
|
||||
"TensorOpSimplifier": 3.100177049636841,
|
||||
"TensorOpTransform": 20.175758361816406,
|
||||
"TensorizerLegalizationPass": 0.000455000001238659,
|
||||
"TileCCOps": 0.17463970184326172,
|
||||
"TilingProfiler": 0.7014575004577637,
|
||||
"TransformConvOp": 0.9923229217529297,
|
||||
"TritiumFusion": 0.33277034759521484,
|
||||
"ValueNumbering": 0.08796954154968262,
|
||||
"VectorizeDMA": 0.520815372467041,
|
||||
"VectorizeMatMult": 0.08555078506469727,
|
||||
"VerifySupportedOps": 0.0002469999890308827,
|
||||
"WeightCoalescing": 0.09020423889160156,
|
||||
"ZeroSizeTensorElimination": 0.0006709098815917969,
|
||||
"algsimp": 0.0018210000125691295,
|
||||
"batchnorm_expander": 0.0022700000554323196,
|
||||
"boundary-marker-removal": 0.0009800000116229057,
|
||||
"call-inliner": 0.0002410000015515834,
|
||||
"canonicalize-boundary-marker": 0.0009430000209249556,
|
||||
"collective-stream-id-checker": 0.0008960000122897327,
|
||||
"comparison-expander": 0.000623999978415668,
|
||||
"computation-deduplicator": 0.003625999903306365,
|
||||
"config-lowering": 0.00017899999511428177,
|
||||
"constant_folding": 0.0668639987707138,
|
||||
"cse": 0.001184999942779541,
|
||||
"dce": 6.399999983841553e-05,
|
||||
"dynamic-slice-transpose": 0.0003530000103637576,
|
||||
"eliminate-redundant-compare": 0.00014400000509340316,
|
||||
"emit-offloaded-dropout": 0.0007820000173524022,
|
||||
"flatten-call-graph": 0.0009519999730400741,
|
||||
"fuse-send-recv": 0.07950299978256226,
|
||||
"hilo-conditional-to-select": 0.00015700000221841037,
|
||||
"hilo::LegalizeAlias": 0.003444999922066927,
|
||||
"hilo::NeuronInstCombine": 0.0008500000112690032,
|
||||
"hilo::NeuronOpFusion": 0.00020799999765586108,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.0007050000131130219,
|
||||
"hilo::ScheduleFusion": 4.3000000005122274e-05,
|
||||
"hilo::SixtyFourHack": 0.0005150000215508044,
|
||||
"hilo::VerifyAliasing": 0.00035099999513477087,
|
||||
"hlo-mac-count": 0.08947300165891647,
|
||||
"io-con-pipe-begin": 6.199999916134402e-05,
|
||||
"io-con-pipe-end": 9.999999974752427e-07,
|
||||
"io-layout-normalization": 0.0706389993429184,
|
||||
"legalize-ccops-for-tensorizer": 0.00013800000306218863,
|
||||
"legalize-compare": 0.00027200000477023423,
|
||||
"lower-argminmax-custom-call": 0.0004130000015720725,
|
||||
"map-inline": 0.0010939999483525753,
|
||||
"metadata-naming": 0.009581999853253365,
|
||||
"mlir::detail::OpToOpPassAdaptor": 0.0004569999873638153,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.5751969814300537,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.00508299982175231,
|
||||
"mlir::mhlo::LowerComplexPass": 0.0012469999492168427,
|
||||
"native-to-custom-softmax": 0.0004349999944679439,
|
||||
"native-to-custom-softmax-dx": 0.00044800000614486635,
|
||||
"neuron-hlo-verifier": 0.3175100088119507,
|
||||
"operand_upcaster": 0.0018560000462457538,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 0.0,
|
||||
"post-partition-simplification": 0.48272499442100525,
|
||||
"pre-hlo-begin": 1.4000000192027073e-05,
|
||||
"pre-hlo-end": 1.9999999949504854e-06,
|
||||
"replace-minimum-constant": 0.00022400000307243317,
|
||||
"reshape-mover": 5.700000110664405e-05,
|
||||
"simplify-concat": 0.002764000091701746,
|
||||
"simplify-while-loops": 9.40000027185306e-05,
|
||||
"transform-variadic-reduce": 0.001218999968841672,
|
||||
"tuple-simplifier": 0.00048099999548867345,
|
||||
"unpack-nested-aws-ntwsr": 0.0007040000054985285,
|
||||
"unroll-while-loop": 1.2000000424450263e-05
|
||||
},
|
||||
"hilo": {
|
||||
"HloMacCount": 1796005888.0,
|
||||
"Traffic": 1252622464.0
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 45730,
|
||||
"StaticProfiler::AifUb": 19.087068557739258,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 29.72186851501465,
|
||||
"StaticProfiler::AverageDmaLength": 2694.496826171875,
|
||||
"StaticProfiler::DDRTransferBytes": 954289776,
|
||||
"StaticProfiler::InternalTransferBytes": 196796000,
|
||||
"StaticProfiler::LoadExpanded": 281500,
|
||||
"StaticProfiler::StoreExpanded": 4273,
|
||||
"StaticProfiler::TotalDMAExpanded": 285773,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 62161,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 53675,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 123,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 32320,
|
||||
"TilingProfiler::NumPfTransposes": 292,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 142,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 8762,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 6564,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 536,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 171,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2465,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"all": {
|
||||
"compiletime": {
|
||||
"CanonicalizeConv": 0.0,
|
||||
"CanonicalizeForTensorizer": 0.0005849999724887311,
|
||||
"Canonicalizer": 0.015991000458598137,
|
||||
"HoistCompute": 9.899999713525176e-05,
|
||||
"IdentifyCrossPassTensors": 0.00011500000255182385,
|
||||
"MemcastMotion": 0.0002789999998640269,
|
||||
"PenguinizeFunctions": 0.0005789999850094318,
|
||||
"PruneFunctions": 0.00018899999849963933,
|
||||
"RemoveOptimizationBarriers": 0.0004290000069886446,
|
||||
"ScatterMotion": 0.005127000156790018,
|
||||
"TensorizerLegalizationPass": 0.000455000001238659,
|
||||
"VerifySupportedOps": 0.0002469999890308827,
|
||||
"algsimp": 0.0018210000125691295,
|
||||
"batchnorm_expander": 0.0022700000554323196,
|
||||
"boundary-marker-removal": 0.0009800000116229057,
|
||||
"call-inliner": 0.0002410000015515834,
|
||||
"canonicalize-boundary-marker": 0.0009430000209249556,
|
||||
"collective-stream-id-checker": 0.0008960000122897327,
|
||||
"comparison-expander": 0.000623999978415668,
|
||||
"computation-deduplicator": 0.003625999903306365,
|
||||
"config-lowering": 0.00017899999511428177,
|
||||
"constant_folding": 0.0668639987707138,
|
||||
"cse": 0.001184999942779541,
|
||||
"dce": 6.399999983841553e-05,
|
||||
"dynamic-slice-transpose": 0.0003530000103637576,
|
||||
"eliminate-redundant-compare": 0.00014400000509340316,
|
||||
"emit-offloaded-dropout": 0.0007820000173524022,
|
||||
"flatten-call-graph": 0.0009519999730400741,
|
||||
"fuse-send-recv": 0.07950299978256226,
|
||||
"hilo-conditional-to-select": 0.00015700000221841037,
|
||||
"hilo::LegalizeAlias": 0.003444999922066927,
|
||||
"hilo::NeuronInstCombine": 0.0008500000112690032,
|
||||
"hilo::NeuronOpFusion": 0.00020799999765586108,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.0007050000131130219,
|
||||
"hilo::ScheduleFusion": 4.3000000005122274e-05,
|
||||
"hilo::SixtyFourHack": 0.0005150000215508044,
|
||||
"hilo::VerifyAliasing": 0.00035099999513477087,
|
||||
"hlo-mac-count": 0.08947300165891647,
|
||||
"io-con-pipe-begin": 6.199999916134402e-05,
|
||||
"io-con-pipe-end": 9.999999974752427e-07,
|
||||
"io-layout-normalization": 0.0706389993429184,
|
||||
"legalize-ccops-for-tensorizer": 0.00013800000306218863,
|
||||
"legalize-compare": 0.00027200000477023423,
|
||||
"lower-argminmax-custom-call": 0.0004130000015720725,
|
||||
"map-inline": 0.0010939999483525753,
|
||||
"metadata-naming": 0.009581999853253365,
|
||||
"mlir::detail::OpToOpPassAdaptor": 0.0004569999873638153,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.5751969814300537,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.00508299982175231,
|
||||
"mlir::mhlo::LowerComplexPass": 0.0012469999492168427,
|
||||
"native-to-custom-softmax": 0.0004349999944679439,
|
||||
"native-to-custom-softmax-dx": 0.00044800000614486635,
|
||||
"neuron-hlo-verifier": 0.3175100088119507,
|
||||
"operand_upcaster": 0.0018560000462457538,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 0.0,
|
||||
"post-partition-simplification": 0.48272499442100525,
|
||||
"pre-hlo-begin": 1.4000000192027073e-05,
|
||||
"pre-hlo-end": 1.9999999949504854e-06,
|
||||
"replace-minimum-constant": 0.00022400000307243317,
|
||||
"reshape-mover": 5.700000110664405e-05,
|
||||
"simplify-concat": 0.002764000091701746,
|
||||
"simplify-while-loops": 9.40000027185306e-05,
|
||||
"transform-variadic-reduce": 0.001218999968841672,
|
||||
"tuple-simplifier": 0.00048099999548867345,
|
||||
"unpack-nested-aws-ntwsr": 0.0007040000054985285,
|
||||
"unroll-while-loop": 1.2000000424450263e-05
|
||||
}
|
||||
},
|
||||
"cumsum": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.0002155303955078125,
|
||||
"DMALocalityOpt": 0.00016641616821289063,
|
||||
"DMAProfiler": 0.0007414817810058594,
|
||||
"DataStreaming": 0.0002970695495605469,
|
||||
"DoNothing": 0.0001289844512939453,
|
||||
"ExpandISAMacro": 0.0005061626434326172,
|
||||
"FactorizeBlkDims": 0.0004284381866455078,
|
||||
"InferPSumTensor": 0.0005972385406494141,
|
||||
"InferSharedMemLoc": 0.0002605915069580078,
|
||||
"InsertCoreBarrier": 0.000263214111328125,
|
||||
"LateLegalizeInst": 0.00037741661071777344,
|
||||
"LateNeuronInstComb": 0.0005505084991455078,
|
||||
"LegalizeSundaAccess": 0.0013835430145263672,
|
||||
"LegalizeType": 0.00022840499877929688,
|
||||
"LowerBroadcast": 0.0002124309539794922,
|
||||
"LowerIntrinsics": 0.0002105236053466797,
|
||||
"LowerTranspose": 0.00021958351135253906,
|
||||
"NeuronInstComb": 0.000583648681640625,
|
||||
"NeuronLICM": 0.00035643577575683594,
|
||||
"NeuronSimplifyPredicates": 0.002285003662109375,
|
||||
"NeuronValueNumbering": 0.0004379749298095703,
|
||||
"SFKVectorizer": 0.0024003982543945313,
|
||||
"SimpleAllReduceTiling": 0.00020265579223632813,
|
||||
"SimplifyNeuronTensor": 0.0004928112030029297,
|
||||
"SpillPSum": 0.00045371055603027344,
|
||||
"WeightCoalescing": 0.0002079010009765625
|
||||
}
|
||||
},
|
||||
"sg00": {
|
||||
"hilo": {
|
||||
"ArithmeticIntensity": 2.867593288421631,
|
||||
"HloMacCount": 1796005888.0,
|
||||
"Traffic": 1252622464.0
|
||||
}
|
||||
},
|
||||
"sg0000": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 6.775559425354004,
|
||||
"AffinePredicateResolution": 0.7124242782592773,
|
||||
"AliasDependencyElimination": 0.0037508010864257813,
|
||||
"AliasDependencyInduction": 17.88080596923828,
|
||||
"AliasDependencyReset": 19.01345443725586,
|
||||
"BFComputeCutting": 0.14321279525756836,
|
||||
"BirCodeGenLoop": 4.049878120422363,
|
||||
"CCOpFusion": 0.9675219058990479,
|
||||
"CanonicalizeDAGForPGTiling": 0.18202924728393555,
|
||||
"CanonicalizeIR": 0.8085732460021973,
|
||||
"CoalesceCCOp": 0.2558913230895996,
|
||||
"CommuteConcat": 0.027615070343017578,
|
||||
"DMALocalityOpt": 0.04517340660095215,
|
||||
"DMAProfiler": 0.08888816833496094,
|
||||
"DMATilingProfiler": 0.09855914115905762,
|
||||
"DataLocalityOpt": 2.8309335708618164,
|
||||
"DataStreaming": 0.29266810417175293,
|
||||
"DeConcat": 0.08213448524475098,
|
||||
"DeadCodeElimination": 0.030428647994995117,
|
||||
"DeadStoreElimination": 1.051145315170288,
|
||||
"DelinearIndices": 0.5529322624206543,
|
||||
"Delinearization": 0.14493751525878906,
|
||||
"DelinearizeSPMD": 0.181105375289917,
|
||||
"DoNothing": 7.724761962890625e-05,
|
||||
"DramToDramTranspose": 0.5176758766174316,
|
||||
"DumpGraphAndMetadata": 0.1749565601348877,
|
||||
"EliminateDivs": 1.6087908744812012,
|
||||
"ExpandBatchNorm": 1.096595048904419,
|
||||
"ExpandISAMacro": 0.08580923080444336,
|
||||
"FactorizeBlkDims": 0.49906015396118164,
|
||||
"FactorizeThreadAxesInFreeDims": 0.08890652656555176,
|
||||
"FlattenMacroLoop": 0.08152222633361816,
|
||||
"GenericAccessSimplifier": 0.02545022964477539,
|
||||
"InferInitValue": 1.2978432178497314,
|
||||
"InferIntrinsicOnCC": 0.7616860866546631,
|
||||
"InferNeuronTensor": 2.6053829193115234,
|
||||
"InferNonlocalTensors": 7.113053321838379,
|
||||
"InferPSumTensor": 4.437821388244629,
|
||||
"InferShardAxis": 12.849608421325684,
|
||||
"InferSharedMemLoc": 0.18901801109313965,
|
||||
"InlineNativeKernels": 0.08442234992980957,
|
||||
"InsertCoreBarrier": 0.1489267349243164,
|
||||
"InsertIOTransposes": 0.8695223331451416,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.49620485305786133,
|
||||
"InsertLocalTransposes": 1.934779167175293,
|
||||
"InsertOffloadedTransposes": 0.23423099517822266,
|
||||
"LICM": 0.1213834285736084,
|
||||
"LateLegalizeInst": 0.2814202308654785,
|
||||
"LateLegalizePostSplit": 0.22411513328552246,
|
||||
"LateLowerReshapeOp": 0.033789873123168945,
|
||||
"LateLowerTensorOp": 4.297260761260986,
|
||||
"LateNeuronInstComb": 1.014098882675171,
|
||||
"LayoutPreprocessing": 1.8572075366973877,
|
||||
"LayoutPreprocessingAndAnalysis": 2.2904410362243652,
|
||||
"LayoutRequirementAnalysis": 0.4237644672393799,
|
||||
"LegalizeCCOpLayout": 1.1935856342315674,
|
||||
"LegalizeOpLevelAlias": 0.7793729305267334,
|
||||
"LegalizePartitionReduce": 0.09432244300842285,
|
||||
"LegalizeSundaAccess": 0.972252607345581,
|
||||
"LegalizeSundaMacro": 0.7198967933654785,
|
||||
"LegalizeType": 0.16599535942077637,
|
||||
"LocalLayoutOpt": 0.6666848659515381,
|
||||
"LoopFusion": 0.29988932609558105,
|
||||
"LoopSplitting": 0.05208778381347656,
|
||||
"LowerBroadcast": 0.07862257957458496,
|
||||
"LowerCCOpBlockAxis": 0.2862672805786133,
|
||||
"LowerComplexBroadcast": 0.09356045722961426,
|
||||
"LowerIntrinsics": 1.3567829132080078,
|
||||
"LowerShardAxis": 0.39397311210632324,
|
||||
"LowerTensorOp": 4.018380165100098,
|
||||
"LowerToSendRecv": 0.27350616455078125,
|
||||
"LowerTranspose": 0.513293981552124,
|
||||
"MacroGeneration": 2.0922586917877197,
|
||||
"MaskPropagation": 0.12857651710510254,
|
||||
"MemcpyElimination": 30.280338287353516,
|
||||
"MutateDataType": 0.03657245635986328,
|
||||
"NeuronAliasDependencyInduction": 0.026292085647583008,
|
||||
"NeuronAliasDependencyReset": 0.03253936767578125,
|
||||
"NeuronInstComb": 0.35875773429870605,
|
||||
"NeuronLICM": 0.29233455657958984,
|
||||
"NeuronLoopFusion": 1.711080551147461,
|
||||
"NeuronLoopInterchange": 0.08134770393371582,
|
||||
"NeuronSimplifier": 0.5104148387908936,
|
||||
"NeuronSimplifyPredicates": 0.2326216697692871,
|
||||
"NeuronValueNumbering": 0.11057472229003906,
|
||||
"OptimizeAliasedCopyChain": 0.40732264518737793,
|
||||
"OptimizeNKIKernels": 1.0493371486663818,
|
||||
"PAGLayoutOpt": 18.23680305480957,
|
||||
"PComputeCutting": 0.5901892185211182,
|
||||
"PGLayoutTilingPipeline": 54.30891036987305,
|
||||
"PGTiling": 10.472368240356445,
|
||||
"PadElimination": 0.0187380313873291,
|
||||
"ParAxesAnnotation": 16.293506622314453,
|
||||
"PartialLoopFusion": 1.5408625602722168,
|
||||
"PartialSimdFusion": 0.8349573612213135,
|
||||
"PerfectLoopNest": 0.07371997833251953,
|
||||
"RecognizeOpIdiom": 0.12681221961975098,
|
||||
"Recompute": 0.007548809051513672,
|
||||
"RelaxPredicates": 0.11768078804016113,
|
||||
"Rematerialization": 0.15080475807189941,
|
||||
"RemoveShardedPartitionAxes": 1.1734652519226074,
|
||||
"ReshapeWeights": 0.022508859634399414,
|
||||
"ResolveAccessConflict": 0.20145153999328613,
|
||||
"ResolveComplicatePredicates": 0.3191838264465332,
|
||||
"RewriteReplicationMatmul": 0.039293527603149414,
|
||||
"RewriteWeights": 0.0773766040802002,
|
||||
"SFKVectorizer": 11.047651290893555,
|
||||
"ShardingPropagationAnalysis": 0.665086030960083,
|
||||
"SimpleAllReduceTiling": 0.07064080238342285,
|
||||
"Simplifier": 0.09160900115966797,
|
||||
"SimplifyMacroPredicates": 0.28479433059692383,
|
||||
"SimplifyNeuronTensor": 0.4839820861816406,
|
||||
"SimplifySlice": 0.02684783935546875,
|
||||
"SimplifyTensor": 0.2644212245941162,
|
||||
"SpillPSum": 1.035245418548584,
|
||||
"SplitAPUnionSets": 1.0491511821746826,
|
||||
"SplitAccGrp": 0.055588722229003906,
|
||||
"StaticProfiler": 0.39172840118408203,
|
||||
"StaticTransposeLocalTensor": 0.7211995124816895,
|
||||
"SundaISel": 1.5357580184936523,
|
||||
"TCTransform": 0.029030561447143555,
|
||||
"TensorInitialization": 0.20336008071899414,
|
||||
"TensorOpSimplifier": 3.100177049636841,
|
||||
"TensorOpTransform": 20.175758361816406,
|
||||
"TileCCOps": 0.17463970184326172,
|
||||
"TilingProfiler": 0.7014575004577637,
|
||||
"TransformConvOp": 0.9923229217529297,
|
||||
"TritiumFusion": 0.33277034759521484,
|
||||
"ValueNumbering": 0.08796954154968262,
|
||||
"VectorizeDMA": 0.520815372467041,
|
||||
"VectorizeMatMult": 0.08555078506469727,
|
||||
"WeightCoalescing": 0.08340644836425781,
|
||||
"ZeroSizeTensorElimination": 0.0006709098815917969
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 45730,
|
||||
"StaticProfiler::AifUb": 19.087068557739258,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 29.72186851501465,
|
||||
"StaticProfiler::AverageDmaLength": 2694.496826171875,
|
||||
"StaticProfiler::AverageFractalPeUtilization": 97.78141784667969,
|
||||
"StaticProfiler::AveragePartitionUtilization": 90.33787536621094,
|
||||
"StaticProfiler::AveragePeUtilization": 81.002685546875,
|
||||
"StaticProfiler::DDRTransferBytes": 954289776,
|
||||
"StaticProfiler::InternalTransferBytes": 196796000,
|
||||
"StaticProfiler::LoadExpanded": 281500,
|
||||
"StaticProfiler::LocalizationEfficiency": 155.71730041503906,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 160.65768432617188,
|
||||
"StaticProfiler::StoreExpanded": 4273,
|
||||
"StaticProfiler::TotalDMAExpanded": 285773,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 62161,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 53675,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 123,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 32320,
|
||||
"TilingProfiler::NumPfTransposes": 292,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 142,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 8762,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 6564,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 536,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 171,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2465,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"topk": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.003664731979370117,
|
||||
"DMALocalityOpt": 0.002711057662963867,
|
||||
"DMAProfiler": 0.003767251968383789,
|
||||
"DataStreaming": 0.0067331790924072266,
|
||||
"DoNothing": 0.00014734268188476563,
|
||||
"ExpandISAMacro": 0.006513833999633789,
|
||||
"FactorizeBlkDims": 0.012669801712036133,
|
||||
"InferPSumTensor": 0.012490510940551758,
|
||||
"InferSharedMemLoc": 0.0029754638671875,
|
||||
"InsertCoreBarrier": 0.00330352783203125,
|
||||
"LateLegalizeInst": 0.007447957992553711,
|
||||
"LateNeuronInstComb": 0.008418560028076172,
|
||||
"LegalizeSundaAccess": 0.02680182456970215,
|
||||
"LegalizeType": 0.010413169860839844,
|
||||
"LowerBroadcast": 0.0033206939697265625,
|
||||
"LowerIntrinsics": 0.003704547882080078,
|
||||
"LowerTranspose": 0.0035905838012695313,
|
||||
"NeuronInstComb": 0.008339166641235352,
|
||||
"NeuronLICM": 0.009230852127075195,
|
||||
"NeuronSimplifyPredicates": 0.006308555603027344,
|
||||
"NeuronValueNumbering": 0.003956317901611328,
|
||||
"SFKVectorizer": 0.03573727607727051,
|
||||
"SimpleAllReduceTiling": 0.0036857128143310547,
|
||||
"SimplifyNeuronTensor": 0.054434776306152344,
|
||||
"SpillPSum": 0.02359294891357422,
|
||||
"WeightCoalescing": 0.0065898895263671875
|
||||
}
|
||||
}
|
||||
}
|
||||
3
token_generation_model/_tp0_bk2/graph.neff
Normal file
3
token_generation_model/_tp0_bk2/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2e9104a1923d9a888389573352849a655d6f7005c5782380d0fa3306dab803f3
|
||||
size 3390464
|
||||
4468
token_generation_model/_tp0_bk2/log-neuron-cc.txt
Normal file
4468
token_generation_model/_tp0_bk2/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
token_generation_model/_tp0_bk2/metaneff.pb
Normal file
3
token_generation_model/_tp0_bk2/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:a4a34ac27ace8c75df52c6f7fb5e0c1736f327638e3ac42ff906c6b52e125a5e
|
||||
size 2464555
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0baed4147c67e64b22e5d725b3e55d70e3e16ed2fe838758927f1cb592ebdc7a
|
||||
size 2550451
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2e9104a1923d9a888389573352849a655d6f7005c5782380d0fa3306dab803f3
|
||||
size 3390464
|
||||
224
token_generation_model/_tp0_bk2/neuron_config.json
Normal file
224
token_generation_model/_tp0_bk2/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 4,
|
||||
"bucket_n_active_tokens": false,
|
||||
"buckets": [
|
||||
512
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 1,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": null,
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": false,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 1,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": [
|
||||
512
|
||||
],
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
token_generation_model/_tp0_bk3/command.txt
Normal file
1
token_generation_model/_tp0_bk3/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_02e08d5fdadfbcb8a569+a6f96e4f.hlo_module.pb --output model.MODULE_02e08d5fdadfbcb8a569+a6f96e4f.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk3/log-neuron-cc.txt"]
|
||||
590
token_generation_model/_tp0_bk3/global_metric_store.json
Normal file
590
token_generation_model/_tp0_bk3/global_metric_store.json
Normal file
@@ -0,0 +1,590 @@
|
||||
{
|
||||
"Average": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 97.88961791992188,
|
||||
"StaticProfiler::AveragePartitionUtilization": 89.79463958740234,
|
||||
"StaticProfiler::AveragePeUtilization": 78.904052734375,
|
||||
"StaticProfiler::LocalizationEfficiency": 146.6905975341797,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 151.06674194335938,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
|
||||
}
|
||||
},
|
||||
"Count": {
|
||||
"tensorizer": {
|
||||
"StaticProfiler::AverageFractalPeUtilization": 1,
|
||||
"StaticProfiler::AveragePartitionUtilization": 1,
|
||||
"StaticProfiler::AveragePeUtilization": 1,
|
||||
"StaticProfiler::LocalizationEfficiency": 1,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
|
||||
}
|
||||
},
|
||||
"Sum": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 6.388453960418701,
|
||||
"AffinePredicateResolution": 0.40378904342651367,
|
||||
"AliasDependencyElimination": 0.016310930252075195,
|
||||
"AliasDependencyInduction": 15.282819747924805,
|
||||
"AliasDependencyReset": 16.618453979492188,
|
||||
"BFComputeCutting": 0.1391315460205078,
|
||||
"BirCodeGenLoop": 4.402980804443359,
|
||||
"CCOpFusion": 0.8881416320800781,
|
||||
"CanonicalizeConv": 4.999999873689376e-06,
|
||||
"CanonicalizeDAGForPGTiling": 0.20721173286437988,
|
||||
"CanonicalizeForTensorizer": 0.00076299998909235,
|
||||
"CanonicalizeIR": 1.028043270111084,
|
||||
"Canonicalizer": 0.08714199811220169,
|
||||
"CoalesceCCOp": 0.2357621192932129,
|
||||
"CommuteConcat": 0.0287783145904541,
|
||||
"DMALocalityOpt": 0.07430553436279297,
|
||||
"DMAProfiler": 0.08057951927185059,
|
||||
"DMATilingProfiler": 0.09187006950378418,
|
||||
"DataLocalityOpt": 2.820284605026245,
|
||||
"DataStreaming": 0.4295690059661865,
|
||||
"DeConcat": 0.06640625,
|
||||
"DeadCodeElimination": 0.053946495056152344,
|
||||
"DeadStoreElimination": 1.0540196895599365,
|
||||
"DelinearIndices": 0.3954615592956543,
|
||||
"Delinearization": 0.13945698738098145,
|
||||
"DelinearizeSPMD": 0.17487859725952148,
|
||||
"DoNothing": 0.0004494190216064453,
|
||||
"DramToDramTranspose": 0.34574389457702637,
|
||||
"DumpGraphAndMetadata": 0.23639750480651855,
|
||||
"EliminateDivs": 1.4790117740631104,
|
||||
"ExpandBatchNorm": 1.123610019683838,
|
||||
"ExpandISAMacro": 0.12250304222106934,
|
||||
"FactorizeBlkDims": 0.49267148971557617,
|
||||
"FactorizeThreadAxesInFreeDims": 0.08273863792419434,
|
||||
"FlattenMacroLoop": 0.08472108840942383,
|
||||
"GenericAccessSimplifier": 0.025115966796875,
|
||||
"HoistCompute": 0.0,
|
||||
"IdentifyCrossPassTensors": 0.0003549999964889139,
|
||||
"InferInitValue": 1.3054986000061035,
|
||||
"InferIntrinsicOnCC": 0.7679235935211182,
|
||||
"InferNeuronTensor": 1.671952724456787,
|
||||
"InferNonlocalTensors": 7.110470771789551,
|
||||
"InferPSumTensor": 4.2829790115356445,
|
||||
"InferShardAxis": 12.786405563354492,
|
||||
"InferSharedMemLoc": 0.16948676109313965,
|
||||
"InlineNativeKernels": 0.05566692352294922,
|
||||
"InsertCoreBarrier": 0.15158963203430176,
|
||||
"InsertIOTransposes": 2.3931386470794678,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.5525217056274414,
|
||||
"InsertLocalTransposes": 0.7429313659667969,
|
||||
"InsertOffloadedTransposes": 0.19022369384765625,
|
||||
"LICM": 0.14770793914794922,
|
||||
"LateLegalizeInst": 0.31015753746032715,
|
||||
"LateLegalizePostSplit": 0.22165155410766602,
|
||||
"LateLowerReshapeOp": 0.03522944450378418,
|
||||
"LateLowerTensorOp": 6.813569068908691,
|
||||
"LateNeuronInstComb": 1.2720398902893066,
|
||||
"LayoutPreprocessing": 1.8736376762390137,
|
||||
"LayoutPreprocessingAndAnalysis": 2.3256783485412598,
|
||||
"LayoutRequirementAnalysis": 0.4448714256286621,
|
||||
"LegalizeCCOpLayout": 1.317772388458252,
|
||||
"LegalizeOpLevelAlias": 0.5106728076934814,
|
||||
"LegalizePartitionReduce": 0.08073091506958008,
|
||||
"LegalizeSundaAccess": 0.9463906288146973,
|
||||
"LegalizeSundaMacro": 0.7451527118682861,
|
||||
"LegalizeType": 0.13950800895690918,
|
||||
"LocalLayoutOpt": 0.6608095169067383,
|
||||
"LoopFusion": 0.3101928234100342,
|
||||
"LoopSplitting": 0.036502838134765625,
|
||||
"LowerBroadcast": 0.06515121459960938,
|
||||
"LowerCCOpBlockAxis": 0.32977890968322754,
|
||||
"LowerComplexBroadcast": 0.15041875839233398,
|
||||
"LowerIntrinsics": 1.0637662410736084,
|
||||
"LowerShardAxis": 0.4116206169128418,
|
||||
"LowerTensorOp": 3.285423755645752,
|
||||
"LowerToSendRecv": 0.25904369354248047,
|
||||
"LowerTranspose": 0.5238301753997803,
|
||||
"MacroGeneration": 2.1192569732666016,
|
||||
"MaskPropagation": 0.14647436141967773,
|
||||
"MemcastMotion": 0.0,
|
||||
"MemcpyElimination": 29.3009090423584,
|
||||
"MutateDataType": 0.037621498107910156,
|
||||
"NeuronAliasDependencyInduction": 0.06313967704772949,
|
||||
"NeuronAliasDependencyReset": 0.08403873443603516,
|
||||
"NeuronInstComb": 0.3617970943450928,
|
||||
"NeuronLICM": 0.2975320816040039,
|
||||
"NeuronLoopFusion": 1.491703987121582,
|
||||
"NeuronLoopInterchange": 0.06900668144226074,
|
||||
"NeuronSimplifier": 0.5315425395965576,
|
||||
"NeuronSimplifyPredicates": 0.2570762634277344,
|
||||
"NeuronValueNumbering": 0.11507940292358398,
|
||||
"OptimizeAliasedCopyChain": 0.6118557453155518,
|
||||
"OptimizeNKIKernels": 1.0786433219909668,
|
||||
"PAGLayoutOpt": 18.570398330688477,
|
||||
"PComputeCutting": 0.6222004890441895,
|
||||
"PGLayoutTilingPipeline": 55.62285614013672,
|
||||
"PGTiling": 10.0595703125,
|
||||
"PadElimination": 0.021188735961914063,
|
||||
"ParAxesAnnotation": 17.818796157836914,
|
||||
"PartialLoopFusion": 1.5088038444519043,
|
||||
"PartialSimdFusion": 0.8573262691497803,
|
||||
"PenguinizeFunctions": 0.000754999986384064,
|
||||
"PerfectLoopNest": 0.07143402099609375,
|
||||
"PruneFunctions": 0.0012410000199452043,
|
||||
"RecognizeOpIdiom": 0.1291654109954834,
|
||||
"Recompute": 0.007833480834960938,
|
||||
"RelaxPredicates": 0.14508056640625,
|
||||
"Rematerialization": 0.17157793045043945,
|
||||
"RemoveOptimizationBarriers": 0.0005370000144466758,
|
||||
"RemoveShardedPartitionAxes": 1.3566563129425049,
|
||||
"ReshapeWeights": 0.022662878036499023,
|
||||
"ResolveAccessConflict": 0.2033238410949707,
|
||||
"ResolveComplicatePredicates": 0.49439358711242676,
|
||||
"RewriteReplicationMatmul": 0.03938794136047363,
|
||||
"RewriteWeights": 0.07809877395629883,
|
||||
"SFKVectorizer": 10.350098609924316,
|
||||
"ScatterMotion": 0.0,
|
||||
"ShardingPropagationAnalysis": 0.7199399471282959,
|
||||
"SimpleAllReduceTiling": 0.07370853424072266,
|
||||
"Simplifier": 0.09367704391479492,
|
||||
"SimplifyMacroPredicates": 0.28262877464294434,
|
||||
"SimplifyNeuronTensor": 0.9568257331848145,
|
||||
"SimplifySlice": 0.027447223663330078,
|
||||
"SimplifyTensor": 0.2734675407409668,
|
||||
"SpillPSum": 1.2486350536346436,
|
||||
"SplitAPUnionSets": 0.8627588748931885,
|
||||
"SplitAccGrp": 0.1022341251373291,
|
||||
"StaticProfiler": 0.4753732681274414,
|
||||
"StaticTransposeLocalTensor": 0.6612956523895264,
|
||||
"SundaISel": 1.6735970973968506,
|
||||
"TCTransform": 0.029105186462402344,
|
||||
"TensorInitialization": 0.1944262981414795,
|
||||
"TensorOpSimplifier": 3.3376035690307617,
|
||||
"TensorOpTransform": 19.926769256591797,
|
||||
"TensorizerLegalizationPass": 0.0006489999941550195,
|
||||
"TileCCOps": 0.1856687068939209,
|
||||
"TilingProfiler": 0.5146362781524658,
|
||||
"TransformConvOp": 1.4806337356567383,
|
||||
"TritiumFusion": 0.2828052043914795,
|
||||
"ValueNumbering": 0.08587479591369629,
|
||||
"VectorizeDMA": 0.5646088123321533,
|
||||
"VectorizeMatMult": 0.08723974227905273,
|
||||
"VerifySupportedOps": 0.0004870000120718032,
|
||||
"WeightCoalescing": 0.06492829322814941,
|
||||
"ZeroSizeTensorElimination": 0.0008580684661865234,
|
||||
"algsimp": 0.0014750000555068254,
|
||||
"batchnorm_expander": 0.0015930000226944685,
|
||||
"boundary-marker-removal": 0.0007229999755509198,
|
||||
"call-inliner": 0.00018899999849963933,
|
||||
"canonicalize-boundary-marker": 0.0011119999689981341,
|
||||
"collective-stream-id-checker": 8.900000102585182e-05,
|
||||
"comparison-expander": 0.0007440000190399587,
|
||||
"computation-deduplicator": 0.0037249999586492777,
|
||||
"config-lowering": 0.06129100173711777,
|
||||
"constant_folding": 0.00012700000661425292,
|
||||
"cse": 0.0007800000021234155,
|
||||
"dce": 5.199999941396527e-05,
|
||||
"dynamic-slice-transpose": 0.0002530000056140125,
|
||||
"eliminate-redundant-compare": 0.0001140000022132881,
|
||||
"emit-offloaded-dropout": 0.0006549999816343188,
|
||||
"flatten-call-graph": 0.0013830000534653664,
|
||||
"fuse-send-recv": 0.07738199830055237,
|
||||
"hilo-conditional-to-select": 0.00014400000509340316,
|
||||
"hilo::LegalizeAlias": 0.004616000223904848,
|
||||
"hilo::NeuronInstCombine": 0.0015979999443516135,
|
||||
"hilo::NeuronOpFusion": 0.0001250000059371814,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.0016670000040903687,
|
||||
"hilo::ScheduleFusion": 9.999999747378752e-06,
|
||||
"hilo::SixtyFourHack": 0.0008529999759048223,
|
||||
"hilo::VerifyAliasing": 0.000291000003926456,
|
||||
"hlo-mac-count": 0.02539600059390068,
|
||||
"io-con-pipe-begin": 7.999999979801942e-06,
|
||||
"io-con-pipe-end": 0.0,
|
||||
"io-layout-normalization": 0.0035389999393373728,
|
||||
"legalize-ccops-for-tensorizer": 0.00014200000441633165,
|
||||
"legalize-compare": 0.0007239999831654131,
|
||||
"lower-argminmax-custom-call": 0.0003229999856557697,
|
||||
"map-inline": 0.0010969999711960554,
|
||||
"metadata-naming": 0.009553000330924988,
|
||||
"mlir::detail::OpToOpPassAdaptor": 9.999999747378752e-05,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.6092489957809448,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.0049149999395012856,
|
||||
"mlir::mhlo::LowerComplexPass": 0.008914999663829803,
|
||||
"native-to-custom-softmax": 0.0010550000006332994,
|
||||
"native-to-custom-softmax-dx": 0.001585999969393015,
|
||||
"neuron-hlo-verifier": 0.22506099939346313,
|
||||
"operand_upcaster": 0.003352999920025468,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 0.0,
|
||||
"post-partition-simplification": 0.37791401147842407,
|
||||
"pre-hlo-begin": 3.999999989900971e-06,
|
||||
"pre-hlo-end": 9.999999974752427e-07,
|
||||
"replace-minimum-constant": 0.00019299999985378236,
|
||||
"reshape-mover": 5.2999999752501026e-05,
|
||||
"simplify-concat": 0.06822799891233444,
|
||||
"simplify-while-loops": 5.0999999075429514e-05,
|
||||
"transform-variadic-reduce": 0.0008849999867379665,
|
||||
"tuple-simplifier": 0.00012099999730708078,
|
||||
"unpack-nested-aws-ntwsr": 0.000615999975707382,
|
||||
"unroll-while-loop": 7.000000096013537e-06
|
||||
},
|
||||
"hilo": {
|
||||
"HloMacCount": 1854726144.0,
|
||||
"Traffic": 1252630656.0
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 51674,
|
||||
"StaticProfiler::AifUb": 21.735599517822266,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 31.884082794189453,
|
||||
"StaticProfiler::AverageDmaLength": 2843.9130859375,
|
||||
"StaticProfiler::DDRTransferBytes": 1013018224,
|
||||
"StaticProfiler::InternalTransferBytes": 226623072,
|
||||
"StaticProfiler::LoadExpanded": 281500,
|
||||
"StaticProfiler::StoreExpanded": 4273,
|
||||
"StaticProfiler::TotalDMAExpanded": 285773,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 67989,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 57540,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 123,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 35008,
|
||||
"TilingProfiler::NumPfTransposes": 292,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 142,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 10670,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 8360,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 648,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 283,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2805,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"all": {
|
||||
"compiletime": {
|
||||
"CanonicalizeConv": 4.999999873689376e-06,
|
||||
"CanonicalizeForTensorizer": 0.00076299998909235,
|
||||
"Canonicalizer": 0.08714199811220169,
|
||||
"HoistCompute": 0.0,
|
||||
"IdentifyCrossPassTensors": 0.0003549999964889139,
|
||||
"MemcastMotion": 0.0,
|
||||
"PenguinizeFunctions": 0.000754999986384064,
|
||||
"PruneFunctions": 0.0012410000199452043,
|
||||
"RemoveOptimizationBarriers": 0.0005370000144466758,
|
||||
"ScatterMotion": 0.0,
|
||||
"TensorizerLegalizationPass": 0.0006489999941550195,
|
||||
"VerifySupportedOps": 0.0004870000120718032,
|
||||
"algsimp": 0.0014750000555068254,
|
||||
"batchnorm_expander": 0.0015930000226944685,
|
||||
"boundary-marker-removal": 0.0007229999755509198,
|
||||
"call-inliner": 0.00018899999849963933,
|
||||
"canonicalize-boundary-marker": 0.0011119999689981341,
|
||||
"collective-stream-id-checker": 8.900000102585182e-05,
|
||||
"comparison-expander": 0.0007440000190399587,
|
||||
"computation-deduplicator": 0.0037249999586492777,
|
||||
"config-lowering": 0.06129100173711777,
|
||||
"constant_folding": 0.00012700000661425292,
|
||||
"cse": 0.0007800000021234155,
|
||||
"dce": 5.199999941396527e-05,
|
||||
"dynamic-slice-transpose": 0.0002530000056140125,
|
||||
"eliminate-redundant-compare": 0.0001140000022132881,
|
||||
"emit-offloaded-dropout": 0.0006549999816343188,
|
||||
"flatten-call-graph": 0.0013830000534653664,
|
||||
"fuse-send-recv": 0.07738199830055237,
|
||||
"hilo-conditional-to-select": 0.00014400000509340316,
|
||||
"hilo::LegalizeAlias": 0.004616000223904848,
|
||||
"hilo::NeuronInstCombine": 0.0015979999443516135,
|
||||
"hilo::NeuronOpFusion": 0.0001250000059371814,
|
||||
"hilo::ReplaceTokenTypeWithU8Pass": 0.0016670000040903687,
|
||||
"hilo::ScheduleFusion": 9.999999747378752e-06,
|
||||
"hilo::SixtyFourHack": 0.0008529999759048223,
|
||||
"hilo::VerifyAliasing": 0.000291000003926456,
|
||||
"hlo-mac-count": 0.02539600059390068,
|
||||
"io-con-pipe-begin": 7.999999979801942e-06,
|
||||
"io-con-pipe-end": 0.0,
|
||||
"io-layout-normalization": 0.0035389999393373728,
|
||||
"legalize-ccops-for-tensorizer": 0.00014200000441633165,
|
||||
"legalize-compare": 0.0007239999831654131,
|
||||
"lower-argminmax-custom-call": 0.0003229999856557697,
|
||||
"map-inline": 0.0010969999711960554,
|
||||
"metadata-naming": 0.009553000330924988,
|
||||
"mlir::detail::OpToOpPassAdaptor": 9.999999747378752e-05,
|
||||
"mlir::hlo::MhloToPyPenguin": 0.6092489957809448,
|
||||
"mlir::mhlo::LowerComplexExtraPass": 0.0049149999395012856,
|
||||
"mlir::mhlo::LowerComplexPass": 0.008914999663829803,
|
||||
"native-to-custom-softmax": 0.0010550000006332994,
|
||||
"native-to-custom-softmax-dx": 0.001585999969393015,
|
||||
"neuron-hlo-verifier": 0.22506099939346313,
|
||||
"operand_upcaster": 0.003352999920025468,
|
||||
"post-par-pipe-begin": 9.999999974752427e-07,
|
||||
"post-par-pipe-end": 0.0,
|
||||
"post-partition-simplification": 0.37791401147842407,
|
||||
"pre-hlo-begin": 3.999999989900971e-06,
|
||||
"pre-hlo-end": 9.999999974752427e-07,
|
||||
"replace-minimum-constant": 0.00019299999985378236,
|
||||
"reshape-mover": 5.2999999752501026e-05,
|
||||
"simplify-concat": 0.06822799891233444,
|
||||
"simplify-while-loops": 5.0999999075429514e-05,
|
||||
"transform-variadic-reduce": 0.0008849999867379665,
|
||||
"tuple-simplifier": 0.00012099999730708078,
|
||||
"unpack-nested-aws-ntwsr": 0.000615999975707382,
|
||||
"unroll-while-loop": 7.000000096013537e-06
|
||||
}
|
||||
},
|
||||
"cumsum": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.00023365020751953125,
|
||||
"DMALocalityOpt": 0.00019097328186035156,
|
||||
"DMAProfiler": 0.0009503364562988281,
|
||||
"DataStreaming": 0.00032329559326171875,
|
||||
"DoNothing": 0.00015044212341308594,
|
||||
"ExpandISAMacro": 0.0005333423614501953,
|
||||
"FactorizeBlkDims": 0.0005185604095458984,
|
||||
"InferPSumTensor": 0.0006201267242431641,
|
||||
"InferSharedMemLoc": 0.0002682209014892578,
|
||||
"InsertCoreBarrier": 0.0003275871276855469,
|
||||
"LateLegalizeInst": 0.0004162788391113281,
|
||||
"LateNeuronInstComb": 0.0005996227264404297,
|
||||
"LegalizeSundaAccess": 0.0016045570373535156,
|
||||
"LegalizeType": 0.0002541542053222656,
|
||||
"LowerBroadcast": 0.00023365020751953125,
|
||||
"LowerIntrinsics": 0.00023651123046875,
|
||||
"LowerTranspose": 0.0002307891845703125,
|
||||
"NeuronInstComb": 0.0006008148193359375,
|
||||
"NeuronLICM": 0.00036597251892089844,
|
||||
"NeuronSimplifyPredicates": 0.002484560012817383,
|
||||
"NeuronValueNumbering": 0.0004477500915527344,
|
||||
"SFKVectorizer": 0.002768278121948242,
|
||||
"SimpleAllReduceTiling": 0.00021409988403320313,
|
||||
"SimplifyNeuronTensor": 0.0005936622619628906,
|
||||
"SpillPSum": 0.0005390644073486328,
|
||||
"WeightCoalescing": 0.00023293495178222656
|
||||
}
|
||||
},
|
||||
"sg00": {
|
||||
"hilo": {
|
||||
"ArithmeticIntensity": 2.961329698562622,
|
||||
"HloMacCount": 1854726144.0,
|
||||
"Traffic": 1252630656.0
|
||||
}
|
||||
},
|
||||
"sg0000": {
|
||||
"compiletime": {
|
||||
"AGOrderingAnalysisPass": 6.388453960418701,
|
||||
"AffinePredicateResolution": 0.40378904342651367,
|
||||
"AliasDependencyElimination": 0.016310930252075195,
|
||||
"AliasDependencyInduction": 15.282819747924805,
|
||||
"AliasDependencyReset": 16.618453979492188,
|
||||
"BFComputeCutting": 0.1391315460205078,
|
||||
"BirCodeGenLoop": 4.402980804443359,
|
||||
"CCOpFusion": 0.8881416320800781,
|
||||
"CanonicalizeDAGForPGTiling": 0.20721173286437988,
|
||||
"CanonicalizeIR": 1.028043270111084,
|
||||
"CoalesceCCOp": 0.23186254501342773,
|
||||
"CommuteConcat": 0.0287783145904541,
|
||||
"DMALocalityOpt": 0.07128620147705078,
|
||||
"DMAProfiler": 0.07605504989624023,
|
||||
"DMATilingProfiler": 0.09187006950378418,
|
||||
"DataLocalityOpt": 2.820284605026245,
|
||||
"DataStreaming": 0.42258405685424805,
|
||||
"DeConcat": 0.06640625,
|
||||
"DeadCodeElimination": 0.053946495056152344,
|
||||
"DeadStoreElimination": 1.0540196895599365,
|
||||
"DelinearIndices": 0.3954615592956543,
|
||||
"Delinearization": 0.13945698738098145,
|
||||
"DelinearizeSPMD": 0.17487859725952148,
|
||||
"DoNothing": 0.00014019012451171875,
|
||||
"DramToDramTranspose": 0.34574389457702637,
|
||||
"DumpGraphAndMetadata": 0.23639750480651855,
|
||||
"EliminateDivs": 1.4790117740631104,
|
||||
"ExpandBatchNorm": 1.123610019683838,
|
||||
"ExpandISAMacro": 0.11814379692077637,
|
||||
"FactorizeBlkDims": 0.4797940254211426,
|
||||
"FactorizeThreadAxesInFreeDims": 0.08273863792419434,
|
||||
"FlattenMacroLoop": 0.08472108840942383,
|
||||
"GenericAccessSimplifier": 0.025115966796875,
|
||||
"InferInitValue": 1.3054986000061035,
|
||||
"InferIntrinsicOnCC": 0.7679235935211182,
|
||||
"InferNeuronTensor": 1.671952724456787,
|
||||
"InferNonlocalTensors": 7.110470771789551,
|
||||
"InferPSumTensor": 4.2699432373046875,
|
||||
"InferShardAxis": 12.786405563354492,
|
||||
"InferSharedMemLoc": 0.16634631156921387,
|
||||
"InlineNativeKernels": 0.05566692352294922,
|
||||
"InsertCoreBarrier": 0.1479203701019287,
|
||||
"InsertIOTransposes": 2.3931386470794678,
|
||||
"InsertImplicitShardAxisBeforeISel": 0.5525217056274414,
|
||||
"InsertLocalTransposes": 0.7429313659667969,
|
||||
"InsertOffloadedTransposes": 0.19022369384765625,
|
||||
"LICM": 0.14770793914794922,
|
||||
"LateLegalizeInst": 0.30222225189208984,
|
||||
"LateLegalizePostSplit": 0.22165155410766602,
|
||||
"LateLowerReshapeOp": 0.03522944450378418,
|
||||
"LateLowerTensorOp": 6.813569068908691,
|
||||
"LateNeuronInstComb": 1.2631995677947998,
|
||||
"LayoutPreprocessing": 1.8736376762390137,
|
||||
"LayoutPreprocessingAndAnalysis": 2.3256783485412598,
|
||||
"LayoutRequirementAnalysis": 0.4448714256286621,
|
||||
"LegalizeCCOpLayout": 1.317772388458252,
|
||||
"LegalizeOpLevelAlias": 0.5106728076934814,
|
||||
"LegalizePartitionReduce": 0.08073091506958008,
|
||||
"LegalizeSundaAccess": 0.9303104877471924,
|
||||
"LegalizeSundaMacro": 0.7451527118682861,
|
||||
"LegalizeType": 0.12957215309143066,
|
||||
"LocalLayoutOpt": 0.6608095169067383,
|
||||
"LoopFusion": 0.3101928234100342,
|
||||
"LoopSplitting": 0.036502838134765625,
|
||||
"LowerBroadcast": 0.06164741516113281,
|
||||
"LowerCCOpBlockAxis": 0.32977890968322754,
|
||||
"LowerComplexBroadcast": 0.15041875839233398,
|
||||
"LowerIntrinsics": 1.0597736835479736,
|
||||
"LowerShardAxis": 0.4116206169128418,
|
||||
"LowerTensorOp": 3.285423755645752,
|
||||
"LowerToSendRecv": 0.25904369354248047,
|
||||
"LowerTranspose": 0.5202181339263916,
|
||||
"MacroGeneration": 2.1192569732666016,
|
||||
"MaskPropagation": 0.14647436141967773,
|
||||
"MemcpyElimination": 29.3009090423584,
|
||||
"MutateDataType": 0.037621498107910156,
|
||||
"NeuronAliasDependencyInduction": 0.06313967704772949,
|
||||
"NeuronAliasDependencyReset": 0.08403873443603516,
|
||||
"NeuronInstComb": 0.3526570796966553,
|
||||
"NeuronLICM": 0.28780031204223633,
|
||||
"NeuronLoopFusion": 1.491703987121582,
|
||||
"NeuronLoopInterchange": 0.06900668144226074,
|
||||
"NeuronSimplifier": 0.5315425395965576,
|
||||
"NeuronSimplifyPredicates": 0.25080394744873047,
|
||||
"NeuronValueNumbering": 0.11057853698730469,
|
||||
"OptimizeAliasedCopyChain": 0.6118557453155518,
|
||||
"OptimizeNKIKernels": 1.0786433219909668,
|
||||
"PAGLayoutOpt": 18.570398330688477,
|
||||
"PComputeCutting": 0.6222004890441895,
|
||||
"PGLayoutTilingPipeline": 55.62285614013672,
|
||||
"PGTiling": 10.0595703125,
|
||||
"PadElimination": 0.021188735961914063,
|
||||
"ParAxesAnnotation": 17.818796157836914,
|
||||
"PartialLoopFusion": 1.5088038444519043,
|
||||
"PartialSimdFusion": 0.8573262691497803,
|
||||
"PerfectLoopNest": 0.07143402099609375,
|
||||
"RecognizeOpIdiom": 0.1291654109954834,
|
||||
"Recompute": 0.007833480834960938,
|
||||
"RelaxPredicates": 0.14508056640625,
|
||||
"Rematerialization": 0.17157793045043945,
|
||||
"RemoveShardedPartitionAxes": 1.3566563129425049,
|
||||
"ReshapeWeights": 0.022662878036499023,
|
||||
"ResolveAccessConflict": 0.2033238410949707,
|
||||
"ResolveComplicatePredicates": 0.49439358711242676,
|
||||
"RewriteReplicationMatmul": 0.03938794136047363,
|
||||
"RewriteWeights": 0.07809877395629883,
|
||||
"SFKVectorizer": 10.313403129577637,
|
||||
"ShardingPropagationAnalysis": 0.7199399471282959,
|
||||
"SimpleAllReduceTiling": 0.06984925270080566,
|
||||
"Simplifier": 0.09367704391479492,
|
||||
"SimplifyMacroPredicates": 0.28262877464294434,
|
||||
"SimplifyNeuronTensor": 0.9048190116882324,
|
||||
"SimplifySlice": 0.027447223663330078,
|
||||
"SimplifyTensor": 0.2734675407409668,
|
||||
"SpillPSum": 1.2256271839141846,
|
||||
"SplitAPUnionSets": 0.8627588748931885,
|
||||
"SplitAccGrp": 0.1022341251373291,
|
||||
"StaticProfiler": 0.4753732681274414,
|
||||
"StaticTransposeLocalTensor": 0.6612956523895264,
|
||||
"SundaISel": 1.6735970973968506,
|
||||
"TCTransform": 0.029105186462402344,
|
||||
"TensorInitialization": 0.1944262981414795,
|
||||
"TensorOpSimplifier": 3.3376035690307617,
|
||||
"TensorOpTransform": 19.926769256591797,
|
||||
"TileCCOps": 0.1856687068939209,
|
||||
"TilingProfiler": 0.5146362781524658,
|
||||
"TransformConvOp": 1.4806337356567383,
|
||||
"TritiumFusion": 0.2828052043914795,
|
||||
"ValueNumbering": 0.08587479591369629,
|
||||
"VectorizeDMA": 0.5646088123321533,
|
||||
"VectorizeMatMult": 0.08723974227905273,
|
||||
"WeightCoalescing": 0.06122708320617676,
|
||||
"ZeroSizeTensorElimination": 0.0008580684661865234
|
||||
},
|
||||
"tensorizer": {
|
||||
"DMATilingProfiler::TotalInstructionsAfterTiling": 51674,
|
||||
"StaticProfiler::AifUb": 21.735599517822266,
|
||||
"StaticProfiler::ArithmeticIntensityTensorizer": 31.884082794189453,
|
||||
"StaticProfiler::AverageDmaLength": 2843.9130859375,
|
||||
"StaticProfiler::AverageFractalPeUtilization": 97.88961791992188,
|
||||
"StaticProfiler::AveragePartitionUtilization": 89.79463958740234,
|
||||
"StaticProfiler::AveragePeUtilization": 78.904052734375,
|
||||
"StaticProfiler::DDRTransferBytes": 1013018224,
|
||||
"StaticProfiler::InternalTransferBytes": 226623072,
|
||||
"StaticProfiler::LoadExpanded": 281500,
|
||||
"StaticProfiler::LocalizationEfficiency": 146.6905975341797,
|
||||
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 151.06674194335938,
|
||||
"StaticProfiler::StoreExpanded": 4273,
|
||||
"StaticProfiler::TotalDMAExpanded": 285773,
|
||||
"StaticProfiler::TotalDynamicInstancesCount": 67989,
|
||||
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 57540,
|
||||
"StaticProfiler::TotalLNCComm": 0,
|
||||
"StaticProfiler::TotalLNCCommTransfer": 0,
|
||||
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
|
||||
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::DmaInstructionsAfterTiling": 0,
|
||||
"TilingProfiler::GenericInstructionsAfterTiling": 123,
|
||||
"TilingProfiler::MatMultInstructionsAfterTiling": 35008,
|
||||
"TilingProfiler::NumPfTransposes": 292,
|
||||
"TilingProfiler::NumPfTransposesForIo": 30,
|
||||
"TilingProfiler::NumPfTransposesForLocal": 142,
|
||||
"TilingProfiler::NumPfTransposesForNonlocal": 120,
|
||||
"TilingProfiler::PfTransposeInstructions": 10670,
|
||||
"TilingProfiler::PfTransposeInstructionsForIo": 8360,
|
||||
"TilingProfiler::PfTransposeInstructionsForLocal": 648,
|
||||
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
|
||||
"TilingProfiler::ReduceInstructionsAfterTiling": 283,
|
||||
"TilingProfiler::SimdInstructionsAfterTiling": 2805,
|
||||
"TilingProfiler::TotalInstructionsAfterTiling": 0,
|
||||
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
|
||||
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
|
||||
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
|
||||
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing": 0,
|
||||
"TransformConvOp::conv2d_column_packing_1": 0,
|
||||
"TransformConvOp::conv2d_column_packing_io10": 0,
|
||||
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
|
||||
}
|
||||
},
|
||||
"topk": {
|
||||
"compiletime": {
|
||||
"CoalesceCCOp": 0.003665924072265625,
|
||||
"DMALocalityOpt": 0.002828359603881836,
|
||||
"DMAProfiler": 0.0035741329193115234,
|
||||
"DataStreaming": 0.006661653518676758,
|
||||
"DoNothing": 0.00015878677368164063,
|
||||
"ExpandISAMacro": 0.0038259029388427734,
|
||||
"FactorizeBlkDims": 0.012358903884887695,
|
||||
"InferPSumTensor": 0.01241612434387207,
|
||||
"InferSharedMemLoc": 0.0028722286224365234,
|
||||
"InsertCoreBarrier": 0.0033416748046875,
|
||||
"LateLegalizeInst": 0.0075190067291259766,
|
||||
"LateNeuronInstComb": 0.008240699768066406,
|
||||
"LegalizeSundaAccess": 0.014475584030151367,
|
||||
"LegalizeType": 0.00968170166015625,
|
||||
"LowerBroadcast": 0.0032701492309570313,
|
||||
"LowerIntrinsics": 0.0037560462951660156,
|
||||
"LowerTranspose": 0.0033812522888183594,
|
||||
"NeuronInstComb": 0.008539199829101563,
|
||||
"NeuronLICM": 0.00936579704284668,
|
||||
"NeuronSimplifyPredicates": 0.0037877559661865234,
|
||||
"NeuronValueNumbering": 0.0040531158447265625,
|
||||
"SFKVectorizer": 0.03392672538757324,
|
||||
"SimpleAllReduceTiling": 0.003645181655883789,
|
||||
"SimplifyNeuronTensor": 0.05141305923461914,
|
||||
"SpillPSum": 0.02246880531311035,
|
||||
"WeightCoalescing": 0.0034682750701904297
|
||||
}
|
||||
}
|
||||
}
|
||||
3
token_generation_model/_tp0_bk3/graph.neff
Normal file
3
token_generation_model/_tp0_bk3/graph.neff
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0c81d96024b571f29492feb3e8f54a86bb85699475d770c8027b26b3b25d6d30
|
||||
size 3605504
|
||||
4471
token_generation_model/_tp0_bk3/log-neuron-cc.txt
Normal file
4471
token_generation_model/_tp0_bk3/log-neuron-cc.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
token_generation_model/_tp0_bk3/metaneff.pb
Normal file
3
token_generation_model/_tp0_bk3/metaneff.pb
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:6f9340d294060545b7fd3affe72e9160857b035623ad5e43d5186704d48fc4cd
|
||||
size 2464555
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:25287d0dc180171fa1e47e9880d832a32168762120f79d5891ed820ee9b4fcda
|
||||
size 2550451
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0c81d96024b571f29492feb3e8f54a86bb85699475d770c8027b26b3b25d6d30
|
||||
size 3605504
|
||||
224
token_generation_model/_tp0_bk3/neuron_config.json
Normal file
224
token_generation_model/_tp0_bk3/neuron_config.json
Normal file
@@ -0,0 +1,224 @@
|
||||
{
|
||||
"_attn_implementation_autoset": false,
|
||||
"_name_or_path": "/models/Qwen3-1.7B/",
|
||||
"add_cross_attention": false,
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"attribute_map": {},
|
||||
"bad_words_ids": null,
|
||||
"begin_suppress_tokens": null,
|
||||
"bos_token_id": 151643,
|
||||
"chunk_size_feed_forward": 0,
|
||||
"cross_attention_hidden_size": null,
|
||||
"decoder_start_token_id": null,
|
||||
"diversity_penalty": 0.0,
|
||||
"do_sample": false,
|
||||
"early_stopping": false,
|
||||
"encoder_no_repeat_ngram_size": 0,
|
||||
"eos_token_id": 151645,
|
||||
"exponential_decay_length_penalty": null,
|
||||
"finetuning_task": null,
|
||||
"forced_bos_token_id": null,
|
||||
"forced_eos_token_id": null,
|
||||
"fused_spec_config": null,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"id2label": {
|
||||
"0": "LABEL_0",
|
||||
"1": "LABEL_1"
|
||||
},
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"is_decoder": false,
|
||||
"is_encoder_decoder": false,
|
||||
"label2id": {
|
||||
"LABEL_0": 0,
|
||||
"LABEL_1": 1
|
||||
},
|
||||
"length_penalty": 1.0,
|
||||
"max_length": 20,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 28,
|
||||
"metadata": null,
|
||||
"min_length": 0,
|
||||
"model_type": "qwen3",
|
||||
"neuron_config": {
|
||||
"activation_quantization_type": null,
|
||||
"allow_input_truncation": false,
|
||||
"apply_seq_ids_mask": false,
|
||||
"async_mode": false,
|
||||
"attention_dp_degree": 1,
|
||||
"attention_dtype": null,
|
||||
"attn_block_cte_nki_kernel_enabled": false,
|
||||
"attn_block_tkg_nki_kernel_cache_update": false,
|
||||
"attn_block_tkg_nki_kernel_cascaded_attention": false,
|
||||
"attn_block_tkg_nki_kernel_enabled": false,
|
||||
"attn_cls": {
|
||||
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
|
||||
"__name__": "NeuronQwen3Attention"
|
||||
},
|
||||
"attn_kernel_enabled": null,
|
||||
"attn_tkg_builtin_kernel_enabled": false,
|
||||
"attn_tkg_nki_kernel_enabled": false,
|
||||
"batch_size": 4,
|
||||
"bucket_n_active_tokens": false,
|
||||
"buckets": [
|
||||
1024
|
||||
],
|
||||
"cast_type": "config",
|
||||
"cc_pipeline_tiling_factor": 1,
|
||||
"chunked_prefill_config": null,
|
||||
"context_encoding_buckets": null,
|
||||
"cp_degree": 1,
|
||||
"ctx_batch_size": 1,
|
||||
"disable_kv_cache_tiling": false,
|
||||
"draft_model_modules_to_not_convert": null,
|
||||
"enable_bucketing": true,
|
||||
"enable_cte_modular_flow": false,
|
||||
"enable_eagle_draft_input_norm": false,
|
||||
"enable_eagle_speculation": false,
|
||||
"enable_fused_speculation": false,
|
||||
"enable_long_context_mode": false,
|
||||
"enable_output_completion_notifications": false,
|
||||
"enable_spill_reload_dge": false,
|
||||
"enable_token_tree": false,
|
||||
"ep_degree": 1,
|
||||
"expert_mlp_nki_kernel_enabled": null,
|
||||
"flash_decoding_enabled": false,
|
||||
"fused_qkv": false,
|
||||
"fused_rmsnorm_skip_gamma": false,
|
||||
"is_block_kv_layout": null,
|
||||
"is_chunked_prefill": false,
|
||||
"is_continuous_batching": true,
|
||||
"is_eagle_draft": false,
|
||||
"is_medusa": false,
|
||||
"is_prefill_stage": false,
|
||||
"is_prefix_caching": false,
|
||||
"k_cache_transposed": false,
|
||||
"kv_cache_batch_size": 4,
|
||||
"kv_cache_padding_size": 0,
|
||||
"kv_cache_quant": false,
|
||||
"kv_cache_tiling": false,
|
||||
"layer_boundary_markers": false,
|
||||
"lm_head_pad": true,
|
||||
"lm_head_pad_alignment_size": 1,
|
||||
"local_ranks_size": 4,
|
||||
"logical_nc_config": 2,
|
||||
"lora_config": null,
|
||||
"max_batch_size": 4,
|
||||
"max_context_length": 2048,
|
||||
"max_length": 2048,
|
||||
"max_new_tokens": null,
|
||||
"medusa_speculation_length": 0,
|
||||
"medusa_tree": null,
|
||||
"mlp_kernel_enabled": false,
|
||||
"mlp_kernel_fuse_residual_add": false,
|
||||
"modules_to_not_convert": null,
|
||||
"moe_fused_nki_kernel_enabled": null,
|
||||
"n_active_tokens": 1,
|
||||
"n_positions": 2048,
|
||||
"num_medusa_heads": 0,
|
||||
"on_cpu": false,
|
||||
"on_device_sampling_config": {
|
||||
"deterministic": false,
|
||||
"do_sample": false,
|
||||
"dynamic": true,
|
||||
"global_topk": 256,
|
||||
"on_device_sampling_config": true,
|
||||
"temperature": 1.0,
|
||||
"top_k": 1,
|
||||
"top_k_kernel_enabled": false,
|
||||
"top_p": 1.0
|
||||
},
|
||||
"output_logits": false,
|
||||
"overrides_torch_dtype": true,
|
||||
"pa_block_size": 2048,
|
||||
"pa_num_blocks": 4,
|
||||
"padding_side": "right",
|
||||
"pp_degree": 1,
|
||||
"prefix_buckets": null,
|
||||
"qk_layernorm": false,
|
||||
"qkv_kernel_enabled": false,
|
||||
"qkv_kernel_fuse_residual_add": false,
|
||||
"qkv_kernel_nbsd_layout": false,
|
||||
"quantization_dtype": "int8",
|
||||
"quantization_type": "per_tensor_symmetric",
|
||||
"quantize_clamp_bound": Infinity,
|
||||
"quantized": false,
|
||||
"quantized_checkpoints_path": null,
|
||||
"quantized_mlp_kernel_enabled": false,
|
||||
"rmsnorm_quantize_kernel_enabled": false,
|
||||
"router_topk_nki_kernel_enabled": null,
|
||||
"rpl_reduce_dtype": null,
|
||||
"save_sharded_checkpoint": true,
|
||||
"scratchpad_page_size": null,
|
||||
"seq_len": 2048,
|
||||
"seq_len_threshold_for_cc_tiling": 16384,
|
||||
"sequence_parallel_enabled": false,
|
||||
"shared_mlp_nki_kernel_enabled": null,
|
||||
"skip_sharding": false,
|
||||
"skip_warmup": false,
|
||||
"spec_batch_size": 4,
|
||||
"speculation_length": 0,
|
||||
"start_rank_id": 0,
|
||||
"strided_context_parallel_kernel_enabled": false,
|
||||
"target": null,
|
||||
"tensor_capture_config": null,
|
||||
"tile_cc": false,
|
||||
"tkg_batch_size": 4,
|
||||
"token_generation_buckets": [
|
||||
1024
|
||||
],
|
||||
"token_tree_config": null,
|
||||
"torch_dtype": "bfloat16",
|
||||
"tp_degree": 4,
|
||||
"vocab_parallel": false,
|
||||
"weight_gather_seq_len_threshold": 32768,
|
||||
"weights_to_skip_layout_optimization": [],
|
||||
"world_size": 4
|
||||
},
|
||||
"no_repeat_ngram_size": 0,
|
||||
"num_attention_heads": 16,
|
||||
"num_beam_groups": 1,
|
||||
"num_beams": 1,
|
||||
"num_cores_per_group": 1,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"num_return_sequences": 1,
|
||||
"output_attentions": false,
|
||||
"output_hidden_states": false,
|
||||
"output_scores": false,
|
||||
"pad_token_id": 0,
|
||||
"prefix": null,
|
||||
"problem_type": null,
|
||||
"pruned_heads": {},
|
||||
"remove_invalid_values": false,
|
||||
"repetition_penalty": 1.0,
|
||||
"return_dict": true,
|
||||
"return_dict_in_generate": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000,
|
||||
"sep_token_id": null,
|
||||
"sliding_window": null,
|
||||
"suppress_tokens": null,
|
||||
"task_specific_params": null,
|
||||
"temperature": 1.0,
|
||||
"tf_legacy_loss": false,
|
||||
"tie_encoder_decoder": false,
|
||||
"tie_word_embeddings": true,
|
||||
"tokenizer_class": null,
|
||||
"top_k": 50,
|
||||
"top_p": 1.0,
|
||||
"torchscript": false,
|
||||
"transformers_version": "4.51.0",
|
||||
"typical_p": 1.0,
|
||||
"use_bfloat16": false,
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
token_generation_model/_tp0_bk4/command.txt
Normal file
1
token_generation_model/_tp0_bk4/command.txt
Normal file
@@ -0,0 +1 @@
|
||||
neuronx-cc compile --framework=XLA model.MODULE_b4b44a47076a167bd9a5+795ee7cc.hlo_module.pb --output model.MODULE_b4b44a47076a167bd9a5+795ee7cc.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35
|
||||
@@ -0,0 +1 @@
|
||||
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk4/log-neuron-cc.txt"]
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user