初始化项目,由ModelHub XC社区提供模型

Model: aws-neuron/Qwen3-1.7B-TP4-BS4-SEQ2048
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-30 21:56:23 +08:00
commit fd57cdfd17
114 changed files with 237628 additions and 0 deletions

59
.gitattributes vendored Normal file
View File

@@ -0,0 +1,59 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
tokenizer.json filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk0/graph.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk0/model.MODULE_bdcbea8455ebae357b4c+541d7181.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk1/graph.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk1/model.MODULE_e7dad336ed1c266a3016+bbc3fa47.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk2/graph.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk2/model.MODULE_f36c9ad51e28c98c9723+c4081b94.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk3/graph.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk3/model.MODULE_0984f4c19a044cc11c2a+1c024d2c.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk4/graph.neff filter=lfs diff=lfs merge=lfs -text
context_encoding_model/_tp0_bk4/model.MODULE_e6b44ff520e6c4333666+e9aa1481.neff filter=lfs diff=lfs merge=lfs -text
layout_opt/graph.neff filter=lfs diff=lfs merge=lfs -text
layout_opt/model/graph.hlo filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk0/graph.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk0/model.MODULE_5e3d0ce8963512f9ea2d+937cd7a2.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk0/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk1/graph.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk1/model.MODULE_f53407701fa4882a24c0+55c11e15.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk2/graph.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk2/model.MODULE_c60e20a111715e620aa2+6bb8acc9.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk3/graph.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk3/model.MODULE_02e08d5fdadfbcb8a569+a6f96e4f.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk4/graph.neff filter=lfs diff=lfs merge=lfs -text
token_generation_model/_tp0_bk4/model.MODULE_b4b44a47076a167bd9a5+795ee7cc.neff filter=lfs diff=lfs merge=lfs -text

202
LICENSE Normal file
View File

@@ -0,0 +1,202 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright 2024 Alibaba Cloud
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.

301
README.md Normal file
View File

@@ -0,0 +1,301 @@
---
library_name: transformers
license: apache-2.0
license_link: https://huggingface.co/Qwen/Qwen3-1.7B/blob/main/LICENSE
pipeline_tag: text-generation
base_model:
- Qwen/Qwen3-1.7B-Base
---
# Qwen3-1.7B
<a href="https://chat.qwen.ai/" target="_blank" style="margin: 2px;">
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
## Qwen3 Highlights
Qwen3 is the latest generation of large language models in Qwen series, offering a comprehensive suite of dense and mixture-of-experts (MoE) models. Built upon extensive training, Qwen3 delivers groundbreaking advancements in reasoning, instruction-following, agent capabilities, and multilingual support, with the following key features:
- **Uniquely support of seamless switching between thinking mode** (for complex logical reasoning, math, and coding) and **non-thinking mode** (for efficient, general-purpose dialogue) **within single model**, ensuring optimal performance across various scenarios.
- **Significantly enhancement in its reasoning capabilities**, surpassing previous QwQ (in thinking mode) and Qwen2.5 instruct models (in non-thinking mode) on mathematics, code generation, and commonsense logical reasoning.
- **Superior human preference alignment**, excelling in creative writing, role-playing, multi-turn dialogues, and instruction following, to deliver a more natural, engaging, and immersive conversational experience.
- **Expertise in agent capabilities**, enabling precise integration with external tools in both thinking and unthinking modes and achieving leading performance among open-source models in complex agent-based tasks.
- **Support of 100+ languages and dialects** with strong capabilities for **multilingual instruction following** and **translation**.
## Model Overview
**Qwen3-1.7B** has the following features:
- Type: Causal Language Models
- Training Stage: Pretraining & Post-training
- Number of Parameters: 1.7B
- Number of Paramaters (Non-Embedding): 1.4B
- Number of Layers: 28
- Number of Attention Heads (GQA): 16 for Q and 8 for KV
- Context Length: 32,768
For more details, including benchmark evaluation, hardware requirements, and inference performance, please refer to our [blog](https://qwenlm.github.io/blog/qwen3/), [GitHub](https://github.com/QwenLM/Qwen3), and [Documentation](https://qwen.readthedocs.io/en/latest/).
> [!TIP]
> If you encounter significant endless repetitions, please refer to the [Best Practices](#best-practices) section for optimal sampling parameters, and set the ``presence_penalty`` to 1.5.
## Quickstart
The code of Qwen3 has been in the latest Hugging Face `transformers` and we advise you to use the latest version of `transformers`.
With `transformers<4.51.0`, you will encounter the following error:
```
KeyError: 'qwen3'
```
The following contains a code snippet illustrating how to use the model generate content based on given inputs.
```python
from transformers import AutoModelForCausalLM, AutoTokenizer
model_name = "Qwen/Qwen3-1.7B"
# load the tokenizer and the model
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForCausalLM.from_pretrained(
model_name,
torch_dtype="auto",
device_map="auto"
)
# prepare the model input
prompt = "Give me a short introduction to large language model."
messages = [
{"role": "user", "content": prompt}
]
text = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=True # Switches between thinking and non-thinking modes. Default is True.
)
model_inputs = tokenizer([text], return_tensors="pt").to(model.device)
# conduct text completion
generated_ids = model.generate(
**model_inputs,
max_new_tokens=32768
)
output_ids = generated_ids[0][len(model_inputs.input_ids[0]):].tolist()
# parsing thinking content
try:
# rindex finding 151668 (</think>)
index = len(output_ids) - output_ids[::-1].index(151668)
except ValueError:
index = 0
thinking_content = tokenizer.decode(output_ids[:index], skip_special_tokens=True).strip("\n")
content = tokenizer.decode(output_ids[index:], skip_special_tokens=True).strip("\n")
print("thinking content:", thinking_content)
print("content:", content)
```
For deployment, you can use `sglang>=0.4.6.post1` or `vllm>=0.8.5` or to create an OpenAI-compatible API endpoint:
- SGLang:
```shell
python -m sglang.launch_server --model-path Qwen/Qwen3-1.7B --reasoning-parser qwen3
```
- vLLM:
```shell
vllm serve Qwen/Qwen3-1.7B --enable-reasoning --reasoning-parser deepseek_r1
```
For local use, applications such as Ollama, LMStudio, MLX-LM, llama.cpp, and KTransformers have also supported Qwen3.
## Switching Between Thinking and Non-Thinking Mode
> [!TIP]
> The `enable_thinking` switch is also available in APIs created by SGLang and vLLM.
> Please refer to our documentation for [SGLang](https://qwen.readthedocs.io/en/latest/deployment/sglang.html#thinking-non-thinking-modes) and [vLLM](https://qwen.readthedocs.io/en/latest/deployment/vllm.html#thinking-non-thinking-modes) users.
### `enable_thinking=True`
By default, Qwen3 has thinking capabilities enabled, similar to QwQ-32B. This means the model will use its reasoning abilities to enhance the quality of generated responses. For example, when explicitly setting `enable_thinking=True` or leaving it as the default value in `tokenizer.apply_chat_template`, the model will engage its thinking mode.
```python
text = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=True # True is the default value for enable_thinking
)
```
In this mode, the model will generate think content wrapped in a `<think>...</think>` block, followed by the final response.
> [!NOTE]
> For thinking mode, use `Temperature=0.6`, `TopP=0.95`, `TopK=20`, and `MinP=0` (the default setting in `generation_config.json`). **DO NOT use greedy decoding**, as it can lead to performance degradation and endless repetitions. For more detailed guidance, please refer to the [Best Practices](#best-practices) section.
### `enable_thinking=False`
We provide a hard switch to strictly disable the model's thinking behavior, aligning its functionality with the previous Qwen2.5-Instruct models. This mode is particularly useful in scenarios where disabling thinking is essential for enhancing efficiency.
```python
text = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=False # Setting enable_thinking=False disables thinking mode
)
```
In this mode, the model will not generate any think content and will not include a `<think>...</think>` block.
> [!NOTE]
> For non-thinking mode, we suggest using `Temperature=0.7`, `TopP=0.8`, `TopK=20`, and `MinP=0`. For more detailed guidance, please refer to the [Best Practices](#best-practices) section.
### Advanced Usage: Switching Between Thinking and Non-Thinking Modes via User Input
We provide a soft switch mechanism that allows users to dynamically control the model's behavior when `enable_thinking=True`. Specifically, you can add `/think` and `/no_think` to user prompts or system messages to switch the model's thinking mode from turn to turn. The model will follow the most recent instruction in multi-turn conversations.
Here is an example of a multi-turn conversation:
```python
from transformers import AutoModelForCausalLM, AutoTokenizer
class QwenChatbot:
def __init__(self, model_name="Qwen/Qwen3-1.7B"):
self.tokenizer = AutoTokenizer.from_pretrained(model_name)
self.model = AutoModelForCausalLM.from_pretrained(model_name)
self.history = []
def generate_response(self, user_input):
messages = self.history + [{"role": "user", "content": user_input}]
text = self.tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True
)
inputs = self.tokenizer(text, return_tensors="pt")
response_ids = self.model.generate(**inputs, max_new_tokens=32768)[0][len(inputs.input_ids[0]):].tolist()
response = self.tokenizer.decode(response_ids, skip_special_tokens=True)
# Update history
self.history.append({"role": "user", "content": user_input})
self.history.append({"role": "assistant", "content": response})
return response
# Example Usage
if __name__ == "__main__":
chatbot = QwenChatbot()
# First input (without /think or /no_think tags, thinking mode is enabled by default)
user_input_1 = "How many r's in strawberries?"
print(f"User: {user_input_1}")
response_1 = chatbot.generate_response(user_input_1)
print(f"Bot: {response_1}")
print("----------------------")
# Second input with /no_think
user_input_2 = "Then, how many r's in blueberries? /no_think"
print(f"User: {user_input_2}")
response_2 = chatbot.generate_response(user_input_2)
print(f"Bot: {response_2}")
print("----------------------")
# Third input with /think
user_input_3 = "Really? /think"
print(f"User: {user_input_3}")
response_3 = chatbot.generate_response(user_input_3)
print(f"Bot: {response_3}")
```
> [!NOTE]
> For API compatibility, when `enable_thinking=True`, regardless of whether the user uses `/think` or `/no_think`, the model will always output a block wrapped in `<think>...</think>`. However, the content inside this block may be empty if thinking is disabled.
> When `enable_thinking=False`, the soft switches are not valid. Regardless of any `/think` or `/no_think` tags input by the user, the model will not generate think content and will not include a `<think>...</think>` block.
## Agentic Use
Qwen3 excels in tool calling capabilities. We recommend using [Qwen-Agent](https://github.com/QwenLM/Qwen-Agent) to make the best use of agentic ability of Qwen3. Qwen-Agent encapsulates tool-calling templates and tool-calling parsers internally, greatly reducing coding complexity.
To define the available tools, you can use the MCP configuration file, use the integrated tool of Qwen-Agent, or integrate other tools by yourself.
```python
from qwen_agent.agents import Assistant
# Define LLM
llm_cfg = {
'model': 'Qwen3-1.7B',
# Use the endpoint provided by Alibaba Model Studio:
# 'model_type': 'qwen_dashscope',
# 'api_key': os.getenv('DASHSCOPE_API_KEY'),
# Use a custom endpoint compatible with OpenAI API:
'model_server': 'http://localhost:8000/v1', # api_base
'api_key': 'EMPTY',
# Other parameters:
# 'generate_cfg': {
# # Add: When the response content is `<think>this is the thought</think>this is the answer;
# # Do not add: When the response has been separated by reasoning_content and content.
# 'thought_in_content': True,
# },
}
# Define Tools
tools = [
{'mcpServers': { # You can specify the MCP configuration file
'time': {
'command': 'uvx',
'args': ['mcp-server-time', '--local-timezone=Asia/Shanghai']
},
"fetch": {
"command": "uvx",
"args": ["mcp-server-fetch"]
}
}
},
'code_interpreter', # Built-in tools
]
# Define Agent
bot = Assistant(llm=llm_cfg, function_list=tools)
# Streaming generation
messages = [{'role': 'user', 'content': 'https://qwenlm.github.io/blog/ Introduce the latest developments of Qwen'}]
for responses in bot.run(messages=messages):
pass
print(responses)
```
## Best Practices
To achieve optimal performance, we recommend the following settings:
1. **Sampling Parameters**:
- For thinking mode (`enable_thinking=True`), use `Temperature=0.6`, `TopP=0.95`, `TopK=20`, and `MinP=0`. **DO NOT use greedy decoding**, as it can lead to performance degradation and endless repetitions.
- For non-thinking mode (`enable_thinking=False`), we suggest using `Temperature=0.7`, `TopP=0.8`, `TopK=20`, and `MinP=0`.
- For supported frameworks, you can adjust the `presence_penalty` parameter between 0 and 2 to reduce endless repetitions. However, using a higher value may occasionally result in language mixing and a slight decrease in model performance.
2. **Adequate Output Length**: We recommend using an output length of 32,768 tokens for most queries. For benchmarking on highly complex problems, such as those found in math and programming competitions, we suggest setting the max output length to 38,912 tokens. This provides the model with sufficient space to generate detailed and comprehensive responses, thereby enhancing its overall performance.
3. **Standardize Output Format**: We recommend using prompts to standardize model outputs when benchmarking.
- **Math Problems**: Include "Please reason step by step, and put your final answer within \boxed{}." in the prompt.
- **Multiple-Choice Questions**: Add the following JSON structure to the prompt to standardize responses: "Please show your choice in the `answer` field with only the choice letter, e.g., `"answer": "C"`."
4. **No Thinking Content in History**: In multi-turn conversations, the historical model output should only include the final output part and does not need to include the thinking content. It is implemented in the provided chat template in Jinja2. However, for frameworks that do not directly use the Jinja2 chat template, it is up to the developers to ensure that the best practice is followed.
### Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen3technicalreport,
title={Qwen3 Technical Report},
author={Qwen Team},
year={2025},
eprint={2505.09388},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.09388},
}
```

30
config.json Normal file
View File

@@ -0,0 +1,30 @@
{
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"bos_token_id": 151643,
"eos_token_id": 151645,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"initializer_range": 0.02,
"intermediate_size": 6144,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"model_type": "qwen3",
"num_attention_heads": 16,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sliding_window": null,
"tie_word_embeddings": true,
"torch_dtype": "bfloat16",
"transformers_version": "4.51.0",
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_bdcbea8455ebae357b4c+541d7181.hlo_module.pb --output model.MODULE_bdcbea8455ebae357b4c+541d7181.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk0/log-neuron-cc.txt"]

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:92d81f134ce62a3b1bcb9c577d19d5ea4c462cc6a79b0b0e499d1560e903979b
size 830464

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ec4b0c42828e50013dc0bd682bdb81c0aa36920c9456fd53a653c18581ad49d1
size 1877151

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5239d1061207e7bd19a6be27bb12fddcd4b60b66cb5ba20fb4ab47a4be30d4c5
size 1961858

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:92d81f134ce62a3b1bcb9c577d19d5ea4c462cc6a79b0b0e499d1560e903979b
size 830464

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 1,
"bucket_n_active_tokens": true,
"buckets": [
128
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 2,
"chunked_prefill_config": null,
"context_encoding_buckets": [
128
],
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": true,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 2048,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": null,
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_e7dad336ed1c266a3016+bbc3fa47.hlo_module.pb --output model.MODULE_e7dad336ed1c266a3016+bbc3fa47.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk1/log-neuron-cc.txt"]

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ca594c9d5f303adb5de3e5aa58b2eca9ac1ed3aff762e1b09620d6474407d8bc
size 861184

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:20425c1f386fac6326c8423f6243a37818ce0b36840edfe0fdf47dc9a74af5f2
size 2194530

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4dbd208cf8cb5cfeef58be408e28cdbf0a23d70a7cd3affd1510ceae9fb95c8b
size 2280924

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ca594c9d5f303adb5de3e5aa58b2eca9ac1ed3aff762e1b09620d6474407d8bc
size 861184

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 1,
"bucket_n_active_tokens": true,
"buckets": [
256
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 2,
"chunked_prefill_config": null,
"context_encoding_buckets": [
256
],
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": true,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 2048,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": null,
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_f36c9ad51e28c98c9723+c4081b94.hlo_module.pb --output model.MODULE_f36c9ad51e28c98c9723+c4081b94.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk2/log-neuron-cc.txt"]

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f7292240129556fa9d8d8412b12947fe2577cd0c6367f047d23e7e4526896cce
size 912384

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3a52b69e5f1f8c87121e293c72dfb7fb9dc3f8050a04452be5e7691addcc3371
size 2280546

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:89ed0e9df247f2125a989d4e0bbc25ce61ee24d187117a143bd2bd583a741239
size 2366940

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f7292240129556fa9d8d8412b12947fe2577cd0c6367f047d23e7e4526896cce
size 912384

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 1,
"bucket_n_active_tokens": true,
"buckets": [
512
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 2,
"chunked_prefill_config": null,
"context_encoding_buckets": [
512
],
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": true,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 2048,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": null,
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_0984f4c19a044cc11c2a+1c024d2c.hlo_module.pb --output model.MODULE_0984f4c19a044cc11c2a+1c024d2c.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk3/log-neuron-cc.txt"]

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:88e20e9aacad85427bc64b564935a053a2d089c921c32c96c0f2363c14b1f8fb
size 1004544

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f4bfe18bacca09ec836a16458059ad676212bf8842fbb89645dbe0b21f90bea0
size 2454034

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:64bf60998b259cee8443492d313f8dd44a9b30e8f7465266150157a423e9164c
size 2540428

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:88e20e9aacad85427bc64b564935a053a2d089c921c32c96c0f2363c14b1f8fb
size 1004544

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 1,
"bucket_n_active_tokens": true,
"buckets": [
1024
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 2,
"chunked_prefill_config": null,
"context_encoding_buckets": [
1024
],
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": true,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 2048,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": null,
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_e6b44ff520e6c4333666+e9aa1481.hlo_module.pb --output model.MODULE_e6b44ff520e6c4333666+e9aa1481.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ' --lnc=2 -O1 '--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "--lnc=2", "-O1", "--internal-hlo2tensorizer-options= --modular-flow-mac-threshold=10 --verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/context_encoding_model/_tp0_bk4/log-neuron-cc.txt"]

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f1666f119851fd3de7f6ea4cb5da22169bcf4b1284acdd2f9f3afdc8e33da8b0
size 1250304

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:352da1073d15c0fbd9355ed2074a8ee0012540534b027fe615c424d66b1ee859
size 2798098

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:91f063d37769930b9be102cb6a84963b790e2ac12eb75d4e6018c2b479cf533d
size 2884492

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f1666f119851fd3de7f6ea4cb5da22169bcf4b1284acdd2f9f3afdc8e33da8b0
size 1250304

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 1,
"bucket_n_active_tokens": true,
"buckets": [
2048
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 2,
"chunked_prefill_config": null,
"context_encoding_buckets": [
2048
],
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": true,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 2048,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": null,
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

13
generation_config.json Normal file
View File

@@ -0,0 +1,13 @@
{
"bos_token_id": 151643,
"do_sample": true,
"eos_token_id": [
151645,
151643
],
"pad_token_id": 151643,
"temperature": 0.6,
"top_k": 20,
"top_p": 0.95,
"transformers_version": "4.51.0"
}

1
layout_opt/command.txt Normal file
View File

@@ -0,0 +1 @@
neuronx-cc compile graph.hlo --framework XLA --target trn2 --output graph.neff --model-type=transformer -O1 --lnc=2 '--internal-hlo2tensorizer-options=--experimental-unsafe-fp8e4m3fn-as-fp8e4m3 --verify-hlo=true' --logfile=log-neuron-cc.txt --verbose=35

3
layout_opt/graph.neff Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8c8d68c65608dd3d8d871b6447ca2ef8a60df383cadfef8cbefdcc0391f53c10
size 1055744

3308
layout_opt/log-neuron-cc.txt Normal file

File diff suppressed because it is too large Load Diff

934
layout_opt/metaneff Normal file
View File

@@ -0,0 +1,934 @@
(
input0<05><> <09>2embed_tokens.weight8
;
input1<04><10>2'layers.0.self_attn.o_proj.o_proj.weight8
=
input2<04><02>2)layers.0.self_attn.qkv_proj.v_proj.weight8
1
input3<02>2layers.0.input_layernorm.weight8
7
input4<02>2%layers.0.self_attn.k_layernorm.weight8
=
input5<04><02>2)layers.0.self_attn.qkv_proj.k_proj.weight8
7
input6<02>2%layers.0.self_attn.q_layernorm.weight8
=
input7<04><04>2)layers.0.self_attn.qkv_proj.q_proj.weight8
1
input8<04><10> 2layers.0.mlp.down_proj.weight8
/
input9<04> <0C>2layers.0.mlp.up_proj.weight8
;
input10<02>2(layers.0.post_attention_layernorm.weight8
2
input11<04> <0C>2layers.0.mlp.gate_proj.weight8
<
input12<04><10>2'layers.1.self_attn.o_proj.o_proj.weight8
>
input13<04><02>2)layers.1.self_attn.qkv_proj.v_proj.weight8
2
input14<02>2layers.1.input_layernorm.weight8
8
input15<02>2%layers.1.self_attn.k_layernorm.weight8
>
input16<04><02>2)layers.1.self_attn.qkv_proj.k_proj.weight8
8
input17<02>2%layers.1.self_attn.q_layernorm.weight8
>
input18<04><04>2)layers.1.self_attn.qkv_proj.q_proj.weight8
2
input19<04><10> 2layers.1.mlp.down_proj.weight8
0
input20<04> <0C>2layers.1.mlp.up_proj.weight8
;
input21<02>2(layers.1.post_attention_layernorm.weight8
2
input22<04> <0C>2layers.1.mlp.gate_proj.weight8
<
input23<04><10>2'layers.2.self_attn.o_proj.o_proj.weight8
>
input24<04><02>2)layers.2.self_attn.qkv_proj.v_proj.weight8
2
input25<02>2layers.2.input_layernorm.weight8
8
input26<02>2%layers.2.self_attn.k_layernorm.weight8
>
input27<04><02>2)layers.2.self_attn.qkv_proj.k_proj.weight8
8
input28<02>2%layers.2.self_attn.q_layernorm.weight8
>
input29<04><04>2)layers.2.self_attn.qkv_proj.q_proj.weight8
2
input30<04><10> 2layers.2.mlp.down_proj.weight8
0
input31<04> <0C>2layers.2.mlp.up_proj.weight8
;
input32<02>2(layers.2.post_attention_layernorm.weight8
2
input33<04> <0C>2layers.2.mlp.gate_proj.weight8
<
input34<04><10>2'layers.3.self_attn.o_proj.o_proj.weight8
>
input35<04><02>2)layers.3.self_attn.qkv_proj.v_proj.weight8
2
input36<02>2layers.3.input_layernorm.weight8
8
input37<02>2%layers.3.self_attn.k_layernorm.weight8
>
input38<04><02>2)layers.3.self_attn.qkv_proj.k_proj.weight8
8
input39<02>2%layers.3.self_attn.q_layernorm.weight8
>
input40<04><04>2)layers.3.self_attn.qkv_proj.q_proj.weight8
2
input41<04><10> 2layers.3.mlp.down_proj.weight8
0
input42<04> <0C>2layers.3.mlp.up_proj.weight8
;
input43<02>2(layers.3.post_attention_layernorm.weight8
2
input44<04> <0C>2layers.3.mlp.gate_proj.weight8
<
input45<04><10>2'layers.4.self_attn.o_proj.o_proj.weight8
>
input46<04><02>2)layers.4.self_attn.qkv_proj.v_proj.weight8
2
input47<02>2layers.4.input_layernorm.weight8
8
input48<02>2%layers.4.self_attn.k_layernorm.weight8
>
input49<04><02>2)layers.4.self_attn.qkv_proj.k_proj.weight8
8
input50<02>2%layers.4.self_attn.q_layernorm.weight8
>
input51<04><04>2)layers.4.self_attn.qkv_proj.q_proj.weight8
2
input52<04><10> 2layers.4.mlp.down_proj.weight8
0
input53<04> <0C>2layers.4.mlp.up_proj.weight8
;
input54<02>2(layers.4.post_attention_layernorm.weight8
2
input55<04> <0C>2layers.4.mlp.gate_proj.weight8
<
input56<04><10>2'layers.5.self_attn.o_proj.o_proj.weight8
>
input57<04><02>2)layers.5.self_attn.qkv_proj.v_proj.weight8
2
input58<02>2layers.5.input_layernorm.weight8
8
input59<02>2%layers.5.self_attn.k_layernorm.weight8
>
input60<04><02>2)layers.5.self_attn.qkv_proj.k_proj.weight8
8
input61<02>2%layers.5.self_attn.q_layernorm.weight8
>
input62<04><04>2)layers.5.self_attn.qkv_proj.q_proj.weight8
2
input63<04><10> 2layers.5.mlp.down_proj.weight8
0
input64<04> <0C>2layers.5.mlp.up_proj.weight8
;
input65<02>2(layers.5.post_attention_layernorm.weight8
2
input66<04> <0C>2layers.5.mlp.gate_proj.weight8
<
input67<04><10>2'layers.6.self_attn.o_proj.o_proj.weight8
>
input68<04><02>2)layers.6.self_attn.qkv_proj.v_proj.weight8
2
input69<02>2layers.6.input_layernorm.weight8
8
input70<02>2%layers.6.self_attn.k_layernorm.weight8
>
input71<04><02>2)layers.6.self_attn.qkv_proj.k_proj.weight8
8
input72<02>2%layers.6.self_attn.q_layernorm.weight8
>
input73<04><04>2)layers.6.self_attn.qkv_proj.q_proj.weight8
2
input74<04><10> 2layers.6.mlp.down_proj.weight8
0
input75<04> <0C>2layers.6.mlp.up_proj.weight8
;
input76<02>2(layers.6.post_attention_layernorm.weight8
2
input77<04> <0C>2layers.6.mlp.gate_proj.weight8
<
input78<04><10>2'layers.7.self_attn.o_proj.o_proj.weight8
>
input79<04><02>2)layers.7.self_attn.qkv_proj.v_proj.weight8
2
input80<02>2layers.7.input_layernorm.weight8
8
input81<02>2%layers.7.self_attn.k_layernorm.weight8
>
input82<04><02>2)layers.7.self_attn.qkv_proj.k_proj.weight8
8
input83<02>2%layers.7.self_attn.q_layernorm.weight8
>
input84<04><04>2)layers.7.self_attn.qkv_proj.q_proj.weight8
2
input85<04><10> 2layers.7.mlp.down_proj.weight8
0
input86<04> <0C>2layers.7.mlp.up_proj.weight8
;
input87<02>2(layers.7.post_attention_layernorm.weight8
2
input88<04> <0C>2layers.7.mlp.gate_proj.weight8
<
input89<04><10>2'layers.8.self_attn.o_proj.o_proj.weight8
>
input90<04><02>2)layers.8.self_attn.qkv_proj.v_proj.weight8
2
input91<02>2layers.8.input_layernorm.weight8
8
input92<02>2%layers.8.self_attn.k_layernorm.weight8
>
input93<04><02>2)layers.8.self_attn.qkv_proj.k_proj.weight8
8
input94<02>2%layers.8.self_attn.q_layernorm.weight8
>
input95<04><04>2)layers.8.self_attn.qkv_proj.q_proj.weight8
2
input96<04><10> 2layers.8.mlp.down_proj.weight8
0
input97<04> <0C>2layers.8.mlp.up_proj.weight8
;
input98<02>2(layers.8.post_attention_layernorm.weight8
2
input99<04> <0C>2layers.8.mlp.gate_proj.weight8
=
input100<04><10>2'layers.9.self_attn.o_proj.o_proj.weight8
?
input101<04><02>2)layers.9.self_attn.qkv_proj.v_proj.weight8
3
input102<02>2layers.9.input_layernorm.weight8
9
input103<02>2%layers.9.self_attn.k_layernorm.weight8
?
input104<04><02>2)layers.9.self_attn.qkv_proj.k_proj.weight8
9
input105<02>2%layers.9.self_attn.q_layernorm.weight8
?
input106<04><04>2)layers.9.self_attn.qkv_proj.q_proj.weight8
3
input107<04><10> 2layers.9.mlp.down_proj.weight8
1
input108<04> <0C>2layers.9.mlp.up_proj.weight8
<
input109<02>2(layers.9.post_attention_layernorm.weight8
3
input110<04> <0C>2layers.9.mlp.gate_proj.weight8
>
input111<04><10>2(layers.10.self_attn.o_proj.o_proj.weight8
@
input112<04><02>2*layers.10.self_attn.qkv_proj.v_proj.weight8
4
input113<02>2 layers.10.input_layernorm.weight8
:
input114<02>2&layers.10.self_attn.k_layernorm.weight8
@
input115<04><02>2*layers.10.self_attn.qkv_proj.k_proj.weight8
:
input116<02>2&layers.10.self_attn.q_layernorm.weight8
@
input117<04><04>2*layers.10.self_attn.qkv_proj.q_proj.weight8
4
input118<04><10> 2layers.10.mlp.down_proj.weight8
2
input119<04> <0C>2layers.10.mlp.up_proj.weight8
=
input120<02>2)layers.10.post_attention_layernorm.weight8
4
input121<04> <0C>2layers.10.mlp.gate_proj.weight8
>
input122<04><10>2(layers.11.self_attn.o_proj.o_proj.weight8
@
input123<04><02>2*layers.11.self_attn.qkv_proj.v_proj.weight8
4
input124<02>2 layers.11.input_layernorm.weight8
:
input125<02>2&layers.11.self_attn.k_layernorm.weight8
@
input126<04><02>2*layers.11.self_attn.qkv_proj.k_proj.weight8
:
input127<02>2&layers.11.self_attn.q_layernorm.weight8
@
input128<04><04>2*layers.11.self_attn.qkv_proj.q_proj.weight8
4
input129<04><10> 2layers.11.mlp.down_proj.weight8
2
input130<04> <0C>2layers.11.mlp.up_proj.weight8
=
input131<02>2)layers.11.post_attention_layernorm.weight8
4
input132<04> <0C>2layers.11.mlp.gate_proj.weight8
>
input133<04><10>2(layers.12.self_attn.o_proj.o_proj.weight8
@
input134<04><02>2*layers.12.self_attn.qkv_proj.v_proj.weight8
4
input135<02>2 layers.12.input_layernorm.weight8
:
input136<02>2&layers.12.self_attn.k_layernorm.weight8
@
input137<04><02>2*layers.12.self_attn.qkv_proj.k_proj.weight8
:
input138<02>2&layers.12.self_attn.q_layernorm.weight8
@
input139<04><04>2*layers.12.self_attn.qkv_proj.q_proj.weight8
4
input140<04><10> 2layers.12.mlp.down_proj.weight8
2
input141<04> <0C>2layers.12.mlp.up_proj.weight8
=
input142<02>2)layers.12.post_attention_layernorm.weight8
4
input143<04> <0C>2layers.12.mlp.gate_proj.weight8
>
input144<04><10>2(layers.13.self_attn.o_proj.o_proj.weight8
@
input145<04><02>2*layers.13.self_attn.qkv_proj.v_proj.weight8
4
input146<02>2 layers.13.input_layernorm.weight8
:
input147<02>2&layers.13.self_attn.k_layernorm.weight8
@
input148<04><02>2*layers.13.self_attn.qkv_proj.k_proj.weight8
:
input149<02>2&layers.13.self_attn.q_layernorm.weight8
@
input150<04><04>2*layers.13.self_attn.qkv_proj.q_proj.weight8
4
input151<04><10> 2layers.13.mlp.down_proj.weight8
2
input152<04> <0C>2layers.13.mlp.up_proj.weight8
=
input153<02>2)layers.13.post_attention_layernorm.weight8
4
input154<04> <0C>2layers.13.mlp.gate_proj.weight8
>
input155<04><10>2(layers.14.self_attn.o_proj.o_proj.weight8
@
input156<04><02>2*layers.14.self_attn.qkv_proj.v_proj.weight8
4
input157<02>2 layers.14.input_layernorm.weight8
:
input158<02>2&layers.14.self_attn.k_layernorm.weight8
@
input159<04><02>2*layers.14.self_attn.qkv_proj.k_proj.weight8
:
input160<02>2&layers.14.self_attn.q_layernorm.weight8
@
input161<04><04>2*layers.14.self_attn.qkv_proj.q_proj.weight8
4
input162<04><10> 2layers.14.mlp.down_proj.weight8
2
input163<04> <0C>2layers.14.mlp.up_proj.weight8
=
input164<02>2)layers.14.post_attention_layernorm.weight8
4
input165<04> <0C>2layers.14.mlp.gate_proj.weight8
>
input166<04><10>2(layers.15.self_attn.o_proj.o_proj.weight8
@
input167<04><02>2*layers.15.self_attn.qkv_proj.v_proj.weight8
4
input168<02>2 layers.15.input_layernorm.weight8
:
input169<02>2&layers.15.self_attn.k_layernorm.weight8
@
input170<04><02>2*layers.15.self_attn.qkv_proj.k_proj.weight8
:
input171<02>2&layers.15.self_attn.q_layernorm.weight8
@
input172<04><04>2*layers.15.self_attn.qkv_proj.q_proj.weight8
4
input173<04><10> 2layers.15.mlp.down_proj.weight8
2
input174<04> <0C>2layers.15.mlp.up_proj.weight8
=
input175<02>2)layers.15.post_attention_layernorm.weight8
4
input176<04> <0C>2layers.15.mlp.gate_proj.weight8
>
input177<04><10>2(layers.16.self_attn.o_proj.o_proj.weight8
@
input178<04><02>2*layers.16.self_attn.qkv_proj.v_proj.weight8
4
input179<02>2 layers.16.input_layernorm.weight8
:
input180<02>2&layers.16.self_attn.k_layernorm.weight8
@
input181<04><02>2*layers.16.self_attn.qkv_proj.k_proj.weight8
:
input182<02>2&layers.16.self_attn.q_layernorm.weight8
@
input183<04><04>2*layers.16.self_attn.qkv_proj.q_proj.weight8
4
input184<04><10> 2layers.16.mlp.down_proj.weight8
2
input185<04> <0C>2layers.16.mlp.up_proj.weight8
=
input186<02>2)layers.16.post_attention_layernorm.weight8
4
input187<04> <0C>2layers.16.mlp.gate_proj.weight8
>
input188<04><10>2(layers.17.self_attn.o_proj.o_proj.weight8
@
input189<04><02>2*layers.17.self_attn.qkv_proj.v_proj.weight8
4
input190<02>2 layers.17.input_layernorm.weight8
:
input191<02>2&layers.17.self_attn.k_layernorm.weight8
@
input192<04><02>2*layers.17.self_attn.qkv_proj.k_proj.weight8
:
input193<02>2&layers.17.self_attn.q_layernorm.weight8
@
input194<04><04>2*layers.17.self_attn.qkv_proj.q_proj.weight8
4
input195<04><10> 2layers.17.mlp.down_proj.weight8
2
input196<04> <0C>2layers.17.mlp.up_proj.weight8
=
input197<02>2)layers.17.post_attention_layernorm.weight8
4
input198<04> <0C>2layers.17.mlp.gate_proj.weight8
>
input199<04><10>2(layers.18.self_attn.o_proj.o_proj.weight8
@
input200<04><02>2*layers.18.self_attn.qkv_proj.v_proj.weight8
4
input201<02>2 layers.18.input_layernorm.weight8
:
input202<02>2&layers.18.self_attn.k_layernorm.weight8
@
input203<04><02>2*layers.18.self_attn.qkv_proj.k_proj.weight8
:
input204<02>2&layers.18.self_attn.q_layernorm.weight8
@
input205<04><04>2*layers.18.self_attn.qkv_proj.q_proj.weight8
4
input206<04><10> 2layers.18.mlp.down_proj.weight8
2
input207<04> <0C>2layers.18.mlp.up_proj.weight8
=
input208<02>2)layers.18.post_attention_layernorm.weight8
4
input209<04> <0C>2layers.18.mlp.gate_proj.weight8
>
input210<04><10>2(layers.19.self_attn.o_proj.o_proj.weight8
@
input211<04><02>2*layers.19.self_attn.qkv_proj.v_proj.weight8
4
input212<02>2 layers.19.input_layernorm.weight8
:
input213<02>2&layers.19.self_attn.k_layernorm.weight8
@
input214<04><02>2*layers.19.self_attn.qkv_proj.k_proj.weight8
:
input215<02>2&layers.19.self_attn.q_layernorm.weight8
@
input216<04><04>2*layers.19.self_attn.qkv_proj.q_proj.weight8
4
input217<04><10> 2layers.19.mlp.down_proj.weight8
2
input218<04> <0C>2layers.19.mlp.up_proj.weight8
=
input219<02>2)layers.19.post_attention_layernorm.weight8
4
input220<04> <0C>2layers.19.mlp.gate_proj.weight8
>
input221<04><10>2(layers.20.self_attn.o_proj.o_proj.weight8
@
input222<04><02>2*layers.20.self_attn.qkv_proj.v_proj.weight8
4
input223<02>2 layers.20.input_layernorm.weight8
:
input224<02>2&layers.20.self_attn.k_layernorm.weight8
@
input225<04><02>2*layers.20.self_attn.qkv_proj.k_proj.weight8
:
input226<02>2&layers.20.self_attn.q_layernorm.weight8
@
input227<04><04>2*layers.20.self_attn.qkv_proj.q_proj.weight8
4
input228<04><10> 2layers.20.mlp.down_proj.weight8
2
input229<04> <0C>2layers.20.mlp.up_proj.weight8
=
input230<02>2)layers.20.post_attention_layernorm.weight8
4
input231<04> <0C>2layers.20.mlp.gate_proj.weight8
>
input232<04><10>2(layers.21.self_attn.o_proj.o_proj.weight8
@
input233<04><02>2*layers.21.self_attn.qkv_proj.v_proj.weight8
4
input234<02>2 layers.21.input_layernorm.weight8
:
input235<02>2&layers.21.self_attn.k_layernorm.weight8
@
input236<04><02>2*layers.21.self_attn.qkv_proj.k_proj.weight8
:
input237<02>2&layers.21.self_attn.q_layernorm.weight8
@
input238<04><04>2*layers.21.self_attn.qkv_proj.q_proj.weight8
4
input239<04><10> 2layers.21.mlp.down_proj.weight8
2
input240<04> <0C>2layers.21.mlp.up_proj.weight8
=
input241<02>2)layers.21.post_attention_layernorm.weight8
4
input242<04> <0C>2layers.21.mlp.gate_proj.weight8
>
input243<04><10>2(layers.22.self_attn.o_proj.o_proj.weight8
@
input244<04><02>2*layers.22.self_attn.qkv_proj.v_proj.weight8
4
input245<02>2 layers.22.input_layernorm.weight8
:
input246<02>2&layers.22.self_attn.k_layernorm.weight8
@
input247<04><02>2*layers.22.self_attn.qkv_proj.k_proj.weight8
:
input248<02>2&layers.22.self_attn.q_layernorm.weight8
@
input249<04><04>2*layers.22.self_attn.qkv_proj.q_proj.weight8
4
input250<04><10> 2layers.22.mlp.down_proj.weight8
2
input251<04> <0C>2layers.22.mlp.up_proj.weight8
=
input252<02>2)layers.22.post_attention_layernorm.weight8
4
input253<04> <0C>2layers.22.mlp.gate_proj.weight8
>
input254<04><10>2(layers.23.self_attn.o_proj.o_proj.weight8
@
input255<04><02>2*layers.23.self_attn.qkv_proj.v_proj.weight8
4
input256<02>2 layers.23.input_layernorm.weight8
:
input257<02>2&layers.23.self_attn.k_layernorm.weight8
@
input258<04><02>2*layers.23.self_attn.qkv_proj.k_proj.weight8
:
input259<02>2&layers.23.self_attn.q_layernorm.weight8
@
input260<04><04>2*layers.23.self_attn.qkv_proj.q_proj.weight8
4
input261<04><10> 2layers.23.mlp.down_proj.weight8
2
input262<04> <0C>2layers.23.mlp.up_proj.weight8
=
input263<02>2)layers.23.post_attention_layernorm.weight8
4
input264<04> <0C>2layers.23.mlp.gate_proj.weight8
>
input265<04><10>2(layers.24.self_attn.o_proj.o_proj.weight8
@
input266<04><02>2*layers.24.self_attn.qkv_proj.v_proj.weight8
4
input267<02>2 layers.24.input_layernorm.weight8
:
input268<02>2&layers.24.self_attn.k_layernorm.weight8
@
input269<04><02>2*layers.24.self_attn.qkv_proj.k_proj.weight8
:
input270<02>2&layers.24.self_attn.q_layernorm.weight8
@
input271<04><04>2*layers.24.self_attn.qkv_proj.q_proj.weight8
4
input272<04><10> 2layers.24.mlp.down_proj.weight8
2
input273<04> <0C>2layers.24.mlp.up_proj.weight8
=
input274<02>2)layers.24.post_attention_layernorm.weight8
4
input275<04> <0C>2layers.24.mlp.gate_proj.weight8
>
input276<04><10>2(layers.25.self_attn.o_proj.o_proj.weight8
@
input277<04><02>2*layers.25.self_attn.qkv_proj.v_proj.weight8
4
input278<02>2 layers.25.input_layernorm.weight8
:
input279<02>2&layers.25.self_attn.k_layernorm.weight8
@
input280<04><02>2*layers.25.self_attn.qkv_proj.k_proj.weight8
:
input281<02>2&layers.25.self_attn.q_layernorm.weight8
@
input282<04><04>2*layers.25.self_attn.qkv_proj.q_proj.weight8
4
input283<04><10> 2layers.25.mlp.down_proj.weight8
2
input284<04> <0C>2layers.25.mlp.up_proj.weight8
=
input285<02>2)layers.25.post_attention_layernorm.weight8
4
input286<04> <0C>2layers.25.mlp.gate_proj.weight8
>
input287<04><10>2(layers.26.self_attn.o_proj.o_proj.weight8
@
input288<04><02>2*layers.26.self_attn.qkv_proj.v_proj.weight8
4
input289<02>2 layers.26.input_layernorm.weight8
:
input290<02>2&layers.26.self_attn.k_layernorm.weight8
@
input291<04><02>2*layers.26.self_attn.qkv_proj.k_proj.weight8
:
input292<02>2&layers.26.self_attn.q_layernorm.weight8
@
input293<04><04>2*layers.26.self_attn.qkv_proj.q_proj.weight8
4
input294<04><10> 2layers.26.mlp.down_proj.weight8
2
input295<04> <0C>2layers.26.mlp.up_proj.weight8
=
input296<02>2)layers.26.post_attention_layernorm.weight8
4
input297<04> <0C>2layers.26.mlp.gate_proj.weight8
>
input298<04><10>2(layers.27.self_attn.o_proj.o_proj.weight8
@
input299<04><02>2*layers.27.self_attn.qkv_proj.v_proj.weight8
4
input300<02>2 layers.27.input_layernorm.weight8
:
input301<02>2&layers.27.self_attn.k_layernorm.weight8
@
input302<04><02>2*layers.27.self_attn.qkv_proj.k_proj.weight8
:
input303<02>2&layers.27.self_attn.q_layernorm.weight8
@
input304<04><04>2*layers.27.self_attn.qkv_proj.q_proj.weight8
4
input305<04><10> 2layers.27.mlp.down_proj.weight8
2
input306<04> <0C>2layers.27.mlp.up_proj.weight8
=
input307<02>2)layers.27.post_attention_layernorm.weight8
4
input308<04> <0C>2layers.27.mlp.gate_proj.weight8
%
input309<05><><02>2lm_head.weight8

input310<02>2 norm.weight8'
output0<05><> <09>2embed_tokens.weight:
output1<04><10>2'layers.0.self_attn.o_proj.o_proj.weight<
output2<04><02>2)layers.0.self_attn.qkv_proj.v_proj.weight0
output3<02>2layers.0.input_layernorm.weight6
output4<02>2%layers.0.self_attn.k_layernorm.weight<
output5<04><02>2)layers.0.self_attn.qkv_proj.k_proj.weight6
output6<02>2%layers.0.self_attn.q_layernorm.weight<
output7<04><04>2)layers.0.self_attn.qkv_proj.q_proj.weight0
output8<04><10> 2layers.0.mlp.down_proj.weight.
output9<04> <0C>2layers.0.mlp.up_proj.weight:
output10<02>2(layers.0.post_attention_layernorm.weight1
output11<04> <0C>2layers.0.mlp.gate_proj.weight;
output12<04><10>2'layers.1.self_attn.o_proj.o_proj.weight=
output13<04><02>2)layers.1.self_attn.qkv_proj.v_proj.weight1
output14<02>2layers.1.input_layernorm.weight7
output15<02>2%layers.1.self_attn.k_layernorm.weight=
output16<04><02>2)layers.1.self_attn.qkv_proj.k_proj.weight7
output17<02>2%layers.1.self_attn.q_layernorm.weight=
output18<04><04>2)layers.1.self_attn.qkv_proj.q_proj.weight1
output19<04><10> 2layers.1.mlp.down_proj.weight/
output20<04> <0C>2layers.1.mlp.up_proj.weight:
output21<02>2(layers.1.post_attention_layernorm.weight1
output22<04> <0C>2layers.1.mlp.gate_proj.weight;
output23<04><10>2'layers.2.self_attn.o_proj.o_proj.weight=
output24<04><02>2)layers.2.self_attn.qkv_proj.v_proj.weight1
output25<02>2layers.2.input_layernorm.weight7
output26<02>2%layers.2.self_attn.k_layernorm.weight=
output27<04><02>2)layers.2.self_attn.qkv_proj.k_proj.weight7
output28<02>2%layers.2.self_attn.q_layernorm.weight=
output29<04><04>2)layers.2.self_attn.qkv_proj.q_proj.weight1
output30<04><10> 2layers.2.mlp.down_proj.weight/
output31<04> <0C>2layers.2.mlp.up_proj.weight:
output32<02>2(layers.2.post_attention_layernorm.weight1
output33<04> <0C>2layers.2.mlp.gate_proj.weight;
output34<04><10>2'layers.3.self_attn.o_proj.o_proj.weight=
output35<04><02>2)layers.3.self_attn.qkv_proj.v_proj.weight1
output36<02>2layers.3.input_layernorm.weight7
output37<02>2%layers.3.self_attn.k_layernorm.weight=
output38<04><02>2)layers.3.self_attn.qkv_proj.k_proj.weight7
output39<02>2%layers.3.self_attn.q_layernorm.weight=
output40<04><04>2)layers.3.self_attn.qkv_proj.q_proj.weight1
output41<04><10> 2layers.3.mlp.down_proj.weight/
output42<04> <0C>2layers.3.mlp.up_proj.weight:
output43<02>2(layers.3.post_attention_layernorm.weight1
output44<04> <0C>2layers.3.mlp.gate_proj.weight;
output45<04><10>2'layers.4.self_attn.o_proj.o_proj.weight=
output46<04><02>2)layers.4.self_attn.qkv_proj.v_proj.weight1
output47<02>2layers.4.input_layernorm.weight7
output48<02>2%layers.4.self_attn.k_layernorm.weight=
output49<04><02>2)layers.4.self_attn.qkv_proj.k_proj.weight7
output50<02>2%layers.4.self_attn.q_layernorm.weight=
output51<04><04>2)layers.4.self_attn.qkv_proj.q_proj.weight1
output52<04><10> 2layers.4.mlp.down_proj.weight/
output53<04> <0C>2layers.4.mlp.up_proj.weight:
output54<02>2(layers.4.post_attention_layernorm.weight1
output55<04> <0C>2layers.4.mlp.gate_proj.weight;
output56<04><10>2'layers.5.self_attn.o_proj.o_proj.weight=
output57<04><02>2)layers.5.self_attn.qkv_proj.v_proj.weight1
output58<02>2layers.5.input_layernorm.weight7
output59<02>2%layers.5.self_attn.k_layernorm.weight=
output60<04><02>2)layers.5.self_attn.qkv_proj.k_proj.weight7
output61<02>2%layers.5.self_attn.q_layernorm.weight=
output62<04><04>2)layers.5.self_attn.qkv_proj.q_proj.weight1
output63<04><10> 2layers.5.mlp.down_proj.weight/
output64<04> <0C>2layers.5.mlp.up_proj.weight:
output65<02>2(layers.5.post_attention_layernorm.weight1
output66<04> <0C>2layers.5.mlp.gate_proj.weight;
output67<04><10>2'layers.6.self_attn.o_proj.o_proj.weight=
output68<04><02>2)layers.6.self_attn.qkv_proj.v_proj.weight1
output69<02>2layers.6.input_layernorm.weight7
output70<02>2%layers.6.self_attn.k_layernorm.weight=
output71<04><02>2)layers.6.self_attn.qkv_proj.k_proj.weight7
output72<02>2%layers.6.self_attn.q_layernorm.weight=
output73<04><04>2)layers.6.self_attn.qkv_proj.q_proj.weight1
output74<04><10> 2layers.6.mlp.down_proj.weight/
output75<04> <0C>2layers.6.mlp.up_proj.weight:
output76<02>2(layers.6.post_attention_layernorm.weight1
output77<04> <0C>2layers.6.mlp.gate_proj.weight;
output78<04><10>2'layers.7.self_attn.o_proj.o_proj.weight=
output79<04><02>2)layers.7.self_attn.qkv_proj.v_proj.weight1
output80<02>2layers.7.input_layernorm.weight7
output81<02>2%layers.7.self_attn.k_layernorm.weight=
output82<04><02>2)layers.7.self_attn.qkv_proj.k_proj.weight7
output83<02>2%layers.7.self_attn.q_layernorm.weight=
output84<04><04>2)layers.7.self_attn.qkv_proj.q_proj.weight1
output85<04><10> 2layers.7.mlp.down_proj.weight/
output86<04> <0C>2layers.7.mlp.up_proj.weight:
output87<02>2(layers.7.post_attention_layernorm.weight1
output88<04> <0C>2layers.7.mlp.gate_proj.weight;
output89<04><10>2'layers.8.self_attn.o_proj.o_proj.weight=
output90<04><02>2)layers.8.self_attn.qkv_proj.v_proj.weight1
output91<02>2layers.8.input_layernorm.weight7
output92<02>2%layers.8.self_attn.k_layernorm.weight=
output93<04><02>2)layers.8.self_attn.qkv_proj.k_proj.weight7
output94<02>2%layers.8.self_attn.q_layernorm.weight=
output95<04><04>2)layers.8.self_attn.qkv_proj.q_proj.weight1
output96<04><10> 2layers.8.mlp.down_proj.weight/
output97<04> <0C>2layers.8.mlp.up_proj.weight:
output98<02>2(layers.8.post_attention_layernorm.weight1
output99<04> <0C>2layers.8.mlp.gate_proj.weight<
output100<04><10>2'layers.9.self_attn.o_proj.o_proj.weight>
output101<04><02>2)layers.9.self_attn.qkv_proj.v_proj.weight2
output102<02>2layers.9.input_layernorm.weight8
output103<02>2%layers.9.self_attn.k_layernorm.weight>
output104<04><02>2)layers.9.self_attn.qkv_proj.k_proj.weight8
output105<02>2%layers.9.self_attn.q_layernorm.weight>
output106<04><04>2)layers.9.self_attn.qkv_proj.q_proj.weight2
output107<04><10> 2layers.9.mlp.down_proj.weight0
output108<04> <0C>2layers.9.mlp.up_proj.weight;
output109<02>2(layers.9.post_attention_layernorm.weight2
output110<04> <0C>2layers.9.mlp.gate_proj.weight=
output111<04><10>2(layers.10.self_attn.o_proj.o_proj.weight?
output112<04><02>2*layers.10.self_attn.qkv_proj.v_proj.weight3
output113<02>2 layers.10.input_layernorm.weight9
output114<02>2&layers.10.self_attn.k_layernorm.weight?
output115<04><02>2*layers.10.self_attn.qkv_proj.k_proj.weight9
output116<02>2&layers.10.self_attn.q_layernorm.weight?
output117<04><04>2*layers.10.self_attn.qkv_proj.q_proj.weight3
output118<04><10> 2layers.10.mlp.down_proj.weight1
output119<04> <0C>2layers.10.mlp.up_proj.weight<
output120<02>2)layers.10.post_attention_layernorm.weight3
output121<04> <0C>2layers.10.mlp.gate_proj.weight=
output122<04><10>2(layers.11.self_attn.o_proj.o_proj.weight?
output123<04><02>2*layers.11.self_attn.qkv_proj.v_proj.weight3
output124<02>2 layers.11.input_layernorm.weight9
output125<02>2&layers.11.self_attn.k_layernorm.weight?
output126<04><02>2*layers.11.self_attn.qkv_proj.k_proj.weight9
output127<02>2&layers.11.self_attn.q_layernorm.weight?
output128<04><04>2*layers.11.self_attn.qkv_proj.q_proj.weight3
output129<04><10> 2layers.11.mlp.down_proj.weight1
output130<04> <0C>2layers.11.mlp.up_proj.weight<
output131<02>2)layers.11.post_attention_layernorm.weight3
output132<04> <0C>2layers.11.mlp.gate_proj.weight=
output133<04><10>2(layers.12.self_attn.o_proj.o_proj.weight?
output134<04><02>2*layers.12.self_attn.qkv_proj.v_proj.weight3
output135<02>2 layers.12.input_layernorm.weight9
output136<02>2&layers.12.self_attn.k_layernorm.weight?
output137<04><02>2*layers.12.self_attn.qkv_proj.k_proj.weight9
output138<02>2&layers.12.self_attn.q_layernorm.weight?
output139<04><04>2*layers.12.self_attn.qkv_proj.q_proj.weight3
output140<04><10> 2layers.12.mlp.down_proj.weight1
output141<04> <0C>2layers.12.mlp.up_proj.weight<
output142<02>2)layers.12.post_attention_layernorm.weight3
output143<04> <0C>2layers.12.mlp.gate_proj.weight=
output144<04><10>2(layers.13.self_attn.o_proj.o_proj.weight?
output145<04><02>2*layers.13.self_attn.qkv_proj.v_proj.weight3
output146<02>2 layers.13.input_layernorm.weight9
output147<02>2&layers.13.self_attn.k_layernorm.weight?
output148<04><02>2*layers.13.self_attn.qkv_proj.k_proj.weight9
output149<02>2&layers.13.self_attn.q_layernorm.weight?
output150<04><04>2*layers.13.self_attn.qkv_proj.q_proj.weight3
output151<04><10> 2layers.13.mlp.down_proj.weight1
output152<04> <0C>2layers.13.mlp.up_proj.weight<
output153<02>2)layers.13.post_attention_layernorm.weight3
output154<04> <0C>2layers.13.mlp.gate_proj.weight=
output155<04><10>2(layers.14.self_attn.o_proj.o_proj.weight?
output156<04><02>2*layers.14.self_attn.qkv_proj.v_proj.weight3
output157<02>2 layers.14.input_layernorm.weight9
output158<02>2&layers.14.self_attn.k_layernorm.weight?
output159<04><02>2*layers.14.self_attn.qkv_proj.k_proj.weight9
output160<02>2&layers.14.self_attn.q_layernorm.weight?
output161<04><04>2*layers.14.self_attn.qkv_proj.q_proj.weight3
output162<04><10> 2layers.14.mlp.down_proj.weight1
output163<04> <0C>2layers.14.mlp.up_proj.weight<
output164<02>2)layers.14.post_attention_layernorm.weight3
output165<04> <0C>2layers.14.mlp.gate_proj.weight=
output166<04><10>2(layers.15.self_attn.o_proj.o_proj.weight?
output167<04><02>2*layers.15.self_attn.qkv_proj.v_proj.weight3
output168<02>2 layers.15.input_layernorm.weight9
output169<02>2&layers.15.self_attn.k_layernorm.weight?
output170<04><02>2*layers.15.self_attn.qkv_proj.k_proj.weight9
output171<02>2&layers.15.self_attn.q_layernorm.weight?
output172<04><04>2*layers.15.self_attn.qkv_proj.q_proj.weight3
output173<04><10> 2layers.15.mlp.down_proj.weight1
output174<04> <0C>2layers.15.mlp.up_proj.weight<
output175<02>2)layers.15.post_attention_layernorm.weight3
output176<04> <0C>2layers.15.mlp.gate_proj.weight=
output177<04><10>2(layers.16.self_attn.o_proj.o_proj.weight?
output178<04><02>2*layers.16.self_attn.qkv_proj.v_proj.weight3
output179<02>2 layers.16.input_layernorm.weight9
output180<02>2&layers.16.self_attn.k_layernorm.weight?
output181<04><02>2*layers.16.self_attn.qkv_proj.k_proj.weight9
output182<02>2&layers.16.self_attn.q_layernorm.weight?
output183<04><04>2*layers.16.self_attn.qkv_proj.q_proj.weight3
output184<04><10> 2layers.16.mlp.down_proj.weight1
output185<04> <0C>2layers.16.mlp.up_proj.weight<
output186<02>2)layers.16.post_attention_layernorm.weight3
output187<04> <0C>2layers.16.mlp.gate_proj.weight=
output188<04><10>2(layers.17.self_attn.o_proj.o_proj.weight?
output189<04><02>2*layers.17.self_attn.qkv_proj.v_proj.weight3
output190<02>2 layers.17.input_layernorm.weight9
output191<02>2&layers.17.self_attn.k_layernorm.weight?
output192<04><02>2*layers.17.self_attn.qkv_proj.k_proj.weight9
output193<02>2&layers.17.self_attn.q_layernorm.weight?
output194<04><04>2*layers.17.self_attn.qkv_proj.q_proj.weight3
output195<04><10> 2layers.17.mlp.down_proj.weight1
output196<04> <0C>2layers.17.mlp.up_proj.weight<
output197<02>2)layers.17.post_attention_layernorm.weight3
output198<04> <0C>2layers.17.mlp.gate_proj.weight=
output199<04><10>2(layers.18.self_attn.o_proj.o_proj.weight?
output200<04><02>2*layers.18.self_attn.qkv_proj.v_proj.weight3
output201<02>2 layers.18.input_layernorm.weight9
output202<02>2&layers.18.self_attn.k_layernorm.weight?
output203<04><02>2*layers.18.self_attn.qkv_proj.k_proj.weight9
output204<02>2&layers.18.self_attn.q_layernorm.weight?
output205<04><04>2*layers.18.self_attn.qkv_proj.q_proj.weight3
output206<04><10> 2layers.18.mlp.down_proj.weight1
output207<04> <0C>2layers.18.mlp.up_proj.weight<
output208<02>2)layers.18.post_attention_layernorm.weight3
output209<04> <0C>2layers.18.mlp.gate_proj.weight=
output210<04><10>2(layers.19.self_attn.o_proj.o_proj.weight?
output211<04><02>2*layers.19.self_attn.qkv_proj.v_proj.weight3
output212<02>2 layers.19.input_layernorm.weight9
output213<02>2&layers.19.self_attn.k_layernorm.weight?
output214<04><02>2*layers.19.self_attn.qkv_proj.k_proj.weight9
output215<02>2&layers.19.self_attn.q_layernorm.weight?
output216<04><04>2*layers.19.self_attn.qkv_proj.q_proj.weight3
output217<04><10> 2layers.19.mlp.down_proj.weight1
output218<04> <0C>2layers.19.mlp.up_proj.weight<
output219<02>2)layers.19.post_attention_layernorm.weight3
output220<04> <0C>2layers.19.mlp.gate_proj.weight=
output221<04><10>2(layers.20.self_attn.o_proj.o_proj.weight?
output222<04><02>2*layers.20.self_attn.qkv_proj.v_proj.weight3
output223<02>2 layers.20.input_layernorm.weight9
output224<02>2&layers.20.self_attn.k_layernorm.weight?
output225<04><02>2*layers.20.self_attn.qkv_proj.k_proj.weight9
output226<02>2&layers.20.self_attn.q_layernorm.weight?
output227<04><04>2*layers.20.self_attn.qkv_proj.q_proj.weight3
output228<04><10> 2layers.20.mlp.down_proj.weight1
output229<04> <0C>2layers.20.mlp.up_proj.weight<
output230<02>2)layers.20.post_attention_layernorm.weight3
output231<04> <0C>2layers.20.mlp.gate_proj.weight=
output232<04><10>2(layers.21.self_attn.o_proj.o_proj.weight?
output233<04><02>2*layers.21.self_attn.qkv_proj.v_proj.weight3
output234<02>2 layers.21.input_layernorm.weight9
output235<02>2&layers.21.self_attn.k_layernorm.weight?
output236<04><02>2*layers.21.self_attn.qkv_proj.k_proj.weight9
output237<02>2&layers.21.self_attn.q_layernorm.weight?
output238<04><04>2*layers.21.self_attn.qkv_proj.q_proj.weight3
output239<04><10> 2layers.21.mlp.down_proj.weight1
output240<04> <0C>2layers.21.mlp.up_proj.weight<
output241<02>2)layers.21.post_attention_layernorm.weight3
output242<04> <0C>2layers.21.mlp.gate_proj.weight=
output243<04><10>2(layers.22.self_attn.o_proj.o_proj.weight?
output244<04><02>2*layers.22.self_attn.qkv_proj.v_proj.weight3
output245<02>2 layers.22.input_layernorm.weight9
output246<02>2&layers.22.self_attn.k_layernorm.weight?
output247<04><02>2*layers.22.self_attn.qkv_proj.k_proj.weight9
output248<02>2&layers.22.self_attn.q_layernorm.weight?
output249<04><04>2*layers.22.self_attn.qkv_proj.q_proj.weight3
output250<04><10> 2layers.22.mlp.down_proj.weight1
output251<04> <0C>2layers.22.mlp.up_proj.weight<
output252<02>2)layers.22.post_attention_layernorm.weight3
output253<04> <0C>2layers.22.mlp.gate_proj.weight=
output254<04><10>2(layers.23.self_attn.o_proj.o_proj.weight?
output255<04><02>2*layers.23.self_attn.qkv_proj.v_proj.weight3
output256<02>2 layers.23.input_layernorm.weight9
output257<02>2&layers.23.self_attn.k_layernorm.weight?
output258<04><02>2*layers.23.self_attn.qkv_proj.k_proj.weight9
output259<02>2&layers.23.self_attn.q_layernorm.weight?
output260<04><04>2*layers.23.self_attn.qkv_proj.q_proj.weight3
output261<04><10> 2layers.23.mlp.down_proj.weight1
output262<04> <0C>2layers.23.mlp.up_proj.weight<
output263<02>2)layers.23.post_attention_layernorm.weight3
output264<04> <0C>2layers.23.mlp.gate_proj.weight=
output265<04><10>2(layers.24.self_attn.o_proj.o_proj.weight?
output266<04><02>2*layers.24.self_attn.qkv_proj.v_proj.weight3
output267<02>2 layers.24.input_layernorm.weight9
output268<02>2&layers.24.self_attn.k_layernorm.weight?
output269<04><02>2*layers.24.self_attn.qkv_proj.k_proj.weight9
output270<02>2&layers.24.self_attn.q_layernorm.weight?
output271<04><04>2*layers.24.self_attn.qkv_proj.q_proj.weight3
output272<04><10> 2layers.24.mlp.down_proj.weight1
output273<04> <0C>2layers.24.mlp.up_proj.weight<
output274<02>2)layers.24.post_attention_layernorm.weight3
output275<04> <0C>2layers.24.mlp.gate_proj.weight=
output276<04><10>2(layers.25.self_attn.o_proj.o_proj.weight?
output277<04><02>2*layers.25.self_attn.qkv_proj.v_proj.weight3
output278<02>2 layers.25.input_layernorm.weight9
output279<02>2&layers.25.self_attn.k_layernorm.weight?
output280<04><02>2*layers.25.self_attn.qkv_proj.k_proj.weight9
output281<02>2&layers.25.self_attn.q_layernorm.weight?
output282<04><04>2*layers.25.self_attn.qkv_proj.q_proj.weight3
output283<04><10> 2layers.25.mlp.down_proj.weight1
output284<04> <0C>2layers.25.mlp.up_proj.weight<
output285<02>2)layers.25.post_attention_layernorm.weight3
output286<04> <0C>2layers.25.mlp.gate_proj.weight=
output287<04><10>2(layers.26.self_attn.o_proj.o_proj.weight?
output288<04><02>2*layers.26.self_attn.qkv_proj.v_proj.weight3
output289<02>2 layers.26.input_layernorm.weight9
output290<02>2&layers.26.self_attn.k_layernorm.weight?
output291<04><02>2*layers.26.self_attn.qkv_proj.k_proj.weight9
output292<02>2&layers.26.self_attn.q_layernorm.weight?
output293<04><04>2*layers.26.self_attn.qkv_proj.q_proj.weight3
output294<04><10> 2layers.26.mlp.down_proj.weight1
output295<04> <0C>2layers.26.mlp.up_proj.weight<
output296<02>2)layers.26.post_attention_layernorm.weight3
output297<04> <0C>2layers.26.mlp.gate_proj.weight=
output298<04><10>2(layers.27.self_attn.o_proj.o_proj.weight?
output299<04><02>2*layers.27.self_attn.qkv_proj.v_proj.weight3
output300<02>2 layers.27.input_layernorm.weight9
output301<02>2&layers.27.self_attn.k_layernorm.weight?
output302<04><02>2*layers.27.self_attn.qkv_proj.k_proj.weight9
output303<02>2&layers.27.self_attn.q_layernorm.weight?
output304<04><04>2*layers.27.self_attn.qkv_proj.q_proj.weight3
output305<04><10> 2layers.27.mlp.down_proj.weight1
output306<04> <0C>2layers.27.mlp.up_proj.weight<
output307<02>2)layers.27.post_attention_layernorm.weight3
output308<04> <0C>2layers.27.mlp.gate_proj.weight$
output309<05><><02>2lm_head.weight
output310<02>2 norm.weight

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b26580ab7144484c2a38bd00978d0df769574a4d80bac429f0f4b37ce9e628b4
size 196610

151388
merges.txt Normal file

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:169ad53ec313c3a34b06c0809216e4fc072cce444a5d4ff2b59690d064130ed5
size 3441185608

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:912becff8d60672aa8628ef08c05898d9adf17c2ad4ae3caf99b065622fdeff9
size 622329984

3
model.pt Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:312d7cdb71a48e347a18a236be5a96c47f1e7f2075c920123708df9b2c06bedf
size 47333643

View File

@@ -0,0 +1,318 @@
{
"metadata": {
"total_size": 4063479808
},
"weight_map": {
"lm_head.weight": "model-00002-of-00002.safetensors",
"model.embed_tokens.weight": "model-00001-of-00002.safetensors",
"model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.0.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.0.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.1.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.10.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.11.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.12.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.13.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.13.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.14.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.15.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.16.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.17.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.18.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.19.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.19.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.2.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.20.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.21.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.21.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.22.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.22.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.23.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.23.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.24.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.24.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.24.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.24.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.25.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.25.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.25.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.25.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.25.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.26.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.26.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.26.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.26.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.27.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.27.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.27.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.27.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.27.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.3.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.4.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.5.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.6.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.7.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.8.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.8.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.8.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
"model.layers.9.self_attn.k_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.self_attn.q_norm.weight": "model-00001-of-00002.safetensors",
"model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
"model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
"model.norm.weight": "model-00001-of-00002.safetensors"
}
}

222
neuron_config.json Normal file
View File

@@ -0,0 +1,222 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 4,
"bucket_n_active_tokens": false,
"buckets": [
2048
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 2,
"chunked_prefill_config": null,
"context_encoding_buckets": null,
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": null,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 2048,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": null,
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": null,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_5e3d0ce8963512f9ea2d+937cd7a2.hlo_module.pb --output model.MODULE_5e3d0ce8963512f9ea2d+937cd7a2.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --enable-internal-neff-wrapper --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]

View File

@@ -0,0 +1,590 @@
{
"Average": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 96.97119140625,
"StaticProfiler::AveragePartitionUtilization": 88.53791809082031,
"StaticProfiler::AveragePeUtilization": 81.30671691894531,
"StaticProfiler::LocalizationEfficiency": 161.8649139404297,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 167.2097930908203,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
}
},
"Count": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 1,
"StaticProfiler::AveragePartitionUtilization": 1,
"StaticProfiler::AveragePeUtilization": 1,
"StaticProfiler::LocalizationEfficiency": 1,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
}
},
"Sum": {
"compiletime": {
"AGOrderingAnalysisPass": 2.377948760986328,
"AffinePredicateResolution": 0.03525829315185547,
"AliasDependencyElimination": 0.0024094581604003906,
"AliasDependencyInduction": 0.2979590892791748,
"AliasDependencyReset": 0.30464816093444824,
"BFComputeCutting": 0.07316970825195313,
"BirCodeGenLoop": 2.482311487197876,
"CCOpFusion": 0.5325038433074951,
"CanonicalizeConv": 0.00023499999952036887,
"CanonicalizeDAGForPGTiling": 0.1513075828552246,
"CanonicalizeForTensorizer": 0.00026500000967644155,
"CanonicalizeIR": 0.04851794242858887,
"Canonicalizer": 0.004513999912887812,
"CoalesceCCOp": 0.16019058227539063,
"CommuteConcat": 0.024104833602905273,
"DMALocalityOpt": 0.037362098693847656,
"DMAProfiler": 0.07356810569763184,
"DMATilingProfiler": 0.09021782875061035,
"DataLocalityOpt": 3.034395694732666,
"DataStreaming": 0.1286299228668213,
"DeConcat": 0.04242873191833496,
"DeadCodeElimination": 0.02465653419494629,
"DeadStoreElimination": 0.8050529956817627,
"DelinearIndices": 0.46034955978393555,
"Delinearization": 0.11224937438964844,
"DelinearizeSPMD": 0.14354729652404785,
"DoNothing": 0.00033402442932128906,
"DramToDramTranspose": 0.2800898551940918,
"DumpGraphAndMetadata": 0.14458084106445313,
"EliminateDivs": 0.11128425598144531,
"ExpandBatchNorm": 0.04912686347961426,
"ExpandISAMacro": 0.07862186431884766,
"FactorizeBlkDims": 0.49788737297058105,
"FactorizeThreadAxesInFreeDims": 0.05206179618835449,
"FlattenMacroLoop": 0.08013129234313965,
"GenericAccessSimplifier": 0.022362232208251953,
"HoistCompute": 3.899999865097925e-05,
"IdentifyCrossPassTensors": 0.00022899999748915434,
"InferInitValue": 1.3044743537902832,
"InferIntrinsicOnCC": 0.26564502716064453,
"InferNeuronTensor": 1.4273626804351807,
"InferNonlocalTensors": 3.070617198944092,
"InferPSumTensor": 1.877610206604004,
"InferShardAxis": 4.046135902404785,
"InferSharedMemLoc": 0.10418868064880371,
"InlineNativeKernels": 0.046288251876831055,
"InsertCoreBarrier": 0.12113451957702637,
"InsertIOTransposes": 1.5809462070465088,
"InsertImplicitShardAxisBeforeISel": 0.37781310081481934,
"InsertLocalTransposes": 0.8543226718902588,
"InsertOffloadedTransposes": 0.0779869556427002,
"LICM": 0.10844612121582031,
"LateLegalizeInst": 0.1468517780303955,
"LateLegalizePostSplit": 0.08665323257446289,
"LateLowerReshapeOp": 0.02942204475402832,
"LateLowerTensorOp": 0.22415471076965332,
"LateNeuronInstComb": 0.9669840335845947,
"LayoutPreprocessing": 0.7032866477966309,
"LayoutPreprocessingAndAnalysis": 1.1142585277557373,
"LayoutRequirementAnalysis": 0.40787601470947266,
"LegalizeCCOpLayout": 0.052059173583984375,
"LegalizeOpLevelAlias": 0.022985219955444336,
"LegalizePartitionReduce": 0.036779165267944336,
"LegalizeSundaAccess": 0.9409115314483643,
"LegalizeSundaMacro": 0.7163646221160889,
"LegalizeType": 0.1376960277557373,
"LocalLayoutOpt": 0.5141463279724121,
"LoopFusion": 0.2369997501373291,
"LoopSplitting": 0.026223182678222656,
"LowerBroadcast": 0.04980111122131348,
"LowerCCOpBlockAxis": 0.17850804328918457,
"LowerComplexBroadcast": 0.06136465072631836,
"LowerIntrinsics": 0.9023723602294922,
"LowerShardAxis": 0.19546747207641602,
"LowerTensorOp": 0.38861823081970215,
"LowerToSendRecv": 0.1479358673095703,
"LowerTranspose": 0.41634607315063477,
"MacroGeneration": 1.907278060913086,
"MaskPropagation": 0.10989689826965332,
"MemcastMotion": 0.00011899999663000926,
"MemcpyElimination": 3.3127715587615967,
"MutateDataType": 0.031205177307128906,
"NeuronAliasDependencyInduction": 0.014397859573364258,
"NeuronAliasDependencyReset": 0.018595218658447266,
"NeuronInstComb": 0.29677605628967285,
"NeuronLICM": 0.2591841220855713,
"NeuronLoopFusion": 0.9968361854553223,
"NeuronLoopInterchange": 0.049851179122924805,
"NeuronSimplifier": 0.4482152462005615,
"NeuronSimplifyPredicates": 0.13344645500183105,
"NeuronValueNumbering": 0.10327720642089844,
"OptimizeAliasedCopyChain": 0.010796785354614258,
"OptimizeNKIKernels": 0.9786763191223145,
"PAGLayoutOpt": 11.095458984375,
"PComputeCutting": 0.28275370597839355,
"PGLayoutTilingPipeline": 27.67894172668457,
"PGTiling": 4.957591533660889,
"PadElimination": 0.009732246398925781,
"ParAxesAnnotation": 10.236833572387695,
"PartialLoopFusion": 1.0972201824188232,
"PartialSimdFusion": 0.529569149017334,
"PenguinizeFunctions": 0.00012799999967683107,
"PerfectLoopNest": 0.059952735900878906,
"PruneFunctions": 0.000391999987186864,
"RecognizeOpIdiom": 0.120391845703125,
"Recompute": 0.005687713623046875,
"RelaxPredicates": 0.09513998031616211,
"Rematerialization": 0.14099407196044922,
"RemoveOptimizationBarriers": 0.0004239999980200082,
"RemoveShardedPartitionAxes": 0.6145329475402832,
"ReshapeWeights": 0.0214080810546875,
"ResolveAccessConflict": 0.1656482219696045,
"ResolveComplicatePredicates": 0.03499007225036621,
"RewriteReplicationMatmul": 0.03796839714050293,
"RewriteWeights": 0.06430387496948242,
"SFKVectorizer": 5.783308029174805,
"ScatterMotion": 0.002827000105753541,
"ShardingPropagationAnalysis": 0.5885381698608398,
"SimpleAllReduceTiling": 0.05830836296081543,
"Simplifier": 0.07847452163696289,
"SimplifyMacroPredicates": 0.25547003746032715,
"SimplifyNeuronTensor": 0.3350214958190918,
"SimplifySlice": 0.02300572395324707,
"SimplifyTensor": 0.2370157241821289,
"SpillPSum": 0.4782223701477051,
"SplitAPUnionSets": 0.6796882152557373,
"SplitAccGrp": 0.03665971755981445,
"StaticProfiler": 0.13864398002624512,
"StaticTransposeLocalTensor": 0.2623753547668457,
"SundaISel": 1.3397884368896484,
"TCTransform": 0.025269269943237305,
"TensorInitialization": 0.16660237312316895,
"TensorOpSimplifier": 0.15259718894958496,
"TensorOpTransform": 0.8282039165496826,
"TensorizerLegalizationPass": 0.00011800000356743112,
"TileCCOps": 0.3066561222076416,
"TilingProfiler": 0.33243393898010254,
"TransformConvOp": 0.05726790428161621,
"TritiumFusion": 0.13151931762695313,
"ValueNumbering": 0.06717801094055176,
"VectorizeDMA": 0.5187788009643555,
"VectorizeMatMult": 0.04601097106933594,
"VerifySupportedOps": 0.000195999993593432,
"WeightCoalescing": 0.05271005630493164,
"ZeroSizeTensorElimination": 0.0003685951232910156,
"algsimp": 0.0011439999798312783,
"batchnorm_expander": 0.0005740000051446259,
"boundary-marker-removal": 0.00017699999443721026,
"call-inliner": 0.00013800000306218863,
"canonicalize-boundary-marker": 0.0003129999968223274,
"collective-stream-id-checker": 7.899999764049426e-05,
"comparison-expander": 0.000195999993593432,
"computation-deduplicator": 0.00031900001340545714,
"config-lowering": 0.00015100000018719584,
"constant_folding": 9.40000027185306e-05,
"cse": 0.00038400001358240843,
"dce": 2.4000000848900527e-05,
"dynamic-slice-transpose": 8.099999831756577e-05,
"eliminate-redundant-compare": 7.999999797903001e-05,
"emit-offloaded-dropout": 0.00013899999612476677,
"flatten-call-graph": 0.0001720000000204891,
"fuse-send-recv": 0.0008440000237897038,
"hilo-conditional-to-select": 5.2999999752501026e-05,
"hilo::LegalizeAlias": 0.0019950000569224358,
"hilo::NeuronInstCombine": 0.0008609999786131084,
"hilo::NeuronOpFusion": 0.0003239999932702631,
"hilo::ReplaceTokenTypeWithU8Pass": 0.00031400000443682075,
"hilo::ScheduleFusion": 2.9000000722589903e-05,
"hilo::SixtyFourHack": 0.00023200000578071922,
"hilo::VerifyAliasing": 5.0999999075429514e-05,
"hlo-mac-count": 0.006806999910622835,
"io-con-pipe-begin": 9.999999747378752e-06,
"io-con-pipe-end": 0.0,
"io-layout-normalization": 0.0014359999913722277,
"legalize-ccops-for-tensorizer": 1.4999999621068127e-05,
"legalize-compare": 0.000155999994603917,
"lower-argminmax-custom-call": 7.599999662488699e-05,
"map-inline": 0.0003760000108741224,
"metadata-naming": 0.0007949999999254942,
"mlir::detail::OpToOpPassAdaptor": 0.00014600000577047467,
"mlir::hlo::MhloToPyPenguin": 0.057739999145269394,
"mlir::mhlo::LowerComplexExtraPass": 0.0017610000213608146,
"mlir::mhlo::LowerComplexPass": 0.00240899994969368,
"native-to-custom-softmax": 0.00022000000171829015,
"native-to-custom-softmax-dx": 0.00023200000578071922,
"neuron-hlo-verifier": 0.01620600000023842,
"operand_upcaster": 0.0005629999795928597,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 0.0,
"post-partition-simplification": 0.03471200168132782,
"pre-hlo-begin": 3.000000106112566e-06,
"pre-hlo-end": 0.0,
"replace-minimum-constant": 0.00010599999950500205,
"reshape-mover": 4.5000000682193786e-05,
"simplify-concat": 0.0008580000139772892,
"simplify-while-loops": 3.300000025774352e-05,
"transform-variadic-reduce": 0.00025599999935366213,
"tuple-simplifier": 8.900000102585182e-05,
"unpack-nested-aws-ntwsr": 0.00014000000373926014,
"unroll-while-loop": 6.000000212225132e-06
},
"hilo": {
"HloMacCount": 1751965696.0,
"Traffic": 1252616320.0
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 41972,
"StaticProfiler::AifUb": 17.11549949645996,
"StaticProfiler::ArithmeticIntensityTensorizer": 27.70398712158203,
"StaticProfiler::AverageDmaLength": 2723.7119140625,
"StaticProfiler::DDRTransferBytes": 918042224,
"StaticProfiler::InternalTransferBytes": 174444896,
"StaticProfiler::LoadExpanded": 238040,
"StaticProfiler::StoreExpanded": 18609,
"StaticProfiler::TotalDMAExpanded": 256649,
"StaticProfiler::TotalDynamicInstancesCount": 57730,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50535,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 179,
"TilingProfiler::MatMultInstructionsAfterTiling": 30304,
"TilingProfiler::NumPfTransposes": 348,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 198,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 7835,
"TilingProfiler::PfTransposeInstructionsForIo": 5217,
"TilingProfiler::PfTransposeInstructionsForLocal": 508,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 2110,
"TilingProfiler::ReduceInstructionsAfterTiling": 59,
"TilingProfiler::SimdInstructionsAfterTiling": 2266,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"all": {
"compiletime": {
"CanonicalizeConv": 0.00023499999952036887,
"CanonicalizeForTensorizer": 0.00026500000967644155,
"Canonicalizer": 0.004513999912887812,
"HoistCompute": 3.899999865097925e-05,
"IdentifyCrossPassTensors": 0.00022899999748915434,
"MemcastMotion": 0.00011899999663000926,
"PenguinizeFunctions": 0.00012799999967683107,
"PruneFunctions": 0.000391999987186864,
"RemoveOptimizationBarriers": 0.0004239999980200082,
"ScatterMotion": 0.002827000105753541,
"TensorizerLegalizationPass": 0.00011800000356743112,
"VerifySupportedOps": 0.000195999993593432,
"algsimp": 0.0011439999798312783,
"batchnorm_expander": 0.0005740000051446259,
"boundary-marker-removal": 0.00017699999443721026,
"call-inliner": 0.00013800000306218863,
"canonicalize-boundary-marker": 0.0003129999968223274,
"collective-stream-id-checker": 7.899999764049426e-05,
"comparison-expander": 0.000195999993593432,
"computation-deduplicator": 0.00031900001340545714,
"config-lowering": 0.00015100000018719584,
"constant_folding": 9.40000027185306e-05,
"cse": 0.00038400001358240843,
"dce": 2.4000000848900527e-05,
"dynamic-slice-transpose": 8.099999831756577e-05,
"eliminate-redundant-compare": 7.999999797903001e-05,
"emit-offloaded-dropout": 0.00013899999612476677,
"flatten-call-graph": 0.0001720000000204891,
"fuse-send-recv": 0.0008440000237897038,
"hilo-conditional-to-select": 5.2999999752501026e-05,
"hilo::LegalizeAlias": 0.0019950000569224358,
"hilo::NeuronInstCombine": 0.0008609999786131084,
"hilo::NeuronOpFusion": 0.0003239999932702631,
"hilo::ReplaceTokenTypeWithU8Pass": 0.00031400000443682075,
"hilo::ScheduleFusion": 2.9000000722589903e-05,
"hilo::SixtyFourHack": 0.00023200000578071922,
"hilo::VerifyAliasing": 5.0999999075429514e-05,
"hlo-mac-count": 0.006806999910622835,
"io-con-pipe-begin": 9.999999747378752e-06,
"io-con-pipe-end": 0.0,
"io-layout-normalization": 0.0014359999913722277,
"legalize-ccops-for-tensorizer": 1.4999999621068127e-05,
"legalize-compare": 0.000155999994603917,
"lower-argminmax-custom-call": 7.599999662488699e-05,
"map-inline": 0.0003760000108741224,
"metadata-naming": 0.0007949999999254942,
"mlir::detail::OpToOpPassAdaptor": 0.00014600000577047467,
"mlir::hlo::MhloToPyPenguin": 0.057739999145269394,
"mlir::mhlo::LowerComplexExtraPass": 0.0017610000213608146,
"mlir::mhlo::LowerComplexPass": 0.00240899994969368,
"native-to-custom-softmax": 0.00022000000171829015,
"native-to-custom-softmax-dx": 0.00023200000578071922,
"neuron-hlo-verifier": 0.01620600000023842,
"operand_upcaster": 0.0005629999795928597,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 0.0,
"post-partition-simplification": 0.03471200168132782,
"pre-hlo-begin": 3.000000106112566e-06,
"pre-hlo-end": 0.0,
"replace-minimum-constant": 0.00010599999950500205,
"reshape-mover": 4.5000000682193786e-05,
"simplify-concat": 0.0008580000139772892,
"simplify-while-loops": 3.300000025774352e-05,
"transform-variadic-reduce": 0.00025599999935366213,
"tuple-simplifier": 8.900000102585182e-05,
"unpack-nested-aws-ntwsr": 0.00014000000373926014,
"unroll-while-loop": 6.000000212225132e-06
}
},
"cumsum": {
"compiletime": {
"CoalesceCCOp": 0.0002205371856689453,
"DMALocalityOpt": 0.00016808509826660156,
"DMAProfiler": 0.0006198883056640625,
"DataStreaming": 0.0002627372741699219,
"DoNothing": 0.00013875961303710938,
"ExpandISAMacro": 0.0005130767822265625,
"FactorizeBlkDims": 0.0005578994750976563,
"InferPSumTensor": 0.0005154609680175781,
"InferSharedMemLoc": 0.00028228759765625,
"InsertCoreBarrier": 0.00024247169494628906,
"LateLegalizeInst": 0.0003752708435058594,
"LateNeuronInstComb": 0.0005624294281005859,
"LegalizeSundaAccess": 0.0013544559478759766,
"LegalizeType": 0.00024437904357910156,
"LowerBroadcast": 0.00022649765014648438,
"LowerIntrinsics": 0.00021767616271972656,
"LowerTranspose": 0.0002415180206298828,
"NeuronInstComb": 0.0007607936859130859,
"NeuronLICM": 0.00041222572326660156,
"NeuronSimplifyPredicates": 0.002053976058959961,
"NeuronValueNumbering": 0.0004031658172607422,
"SFKVectorizer": 0.002403736114501953,
"SimpleAllReduceTiling": 0.0002071857452392578,
"SimplifyNeuronTensor": 0.0004944801330566406,
"SpillPSum": 0.0004985332489013672,
"WeightCoalescing": 0.000213623046875
}
},
"sg00": {
"hilo": {
"ArithmeticIntensity": 2.797290325164795,
"HloMacCount": 1751965696.0,
"Traffic": 1252616320.0
}
},
"sg0000": {
"compiletime": {
"AGOrderingAnalysisPass": 2.377948760986328,
"AffinePredicateResolution": 0.03525829315185547,
"AliasDependencyElimination": 0.0024094581604003906,
"AliasDependencyInduction": 0.2979590892791748,
"AliasDependencyReset": 0.30464816093444824,
"BFComputeCutting": 0.07316970825195313,
"BirCodeGenLoop": 2.482311487197876,
"CCOpFusion": 0.5325038433074951,
"CanonicalizeDAGForPGTiling": 0.1513075828552246,
"CanonicalizeIR": 0.04851794242858887,
"CoalesceCCOp": 0.1563706398010254,
"CommuteConcat": 0.024104833602905273,
"DMALocalityOpt": 0.03418087959289551,
"DMAProfiler": 0.06951689720153809,
"DMATilingProfiler": 0.09021782875061035,
"DataLocalityOpt": 3.034395694732666,
"DataStreaming": 0.12200498580932617,
"DeConcat": 0.04242873191833496,
"DeadCodeElimination": 0.02465653419494629,
"DeadStoreElimination": 0.8050529956817627,
"DelinearIndices": 0.46034955978393555,
"Delinearization": 0.11224937438964844,
"DelinearizeSPMD": 0.14354729652404785,
"DoNothing": 6.198883056640625e-05,
"DramToDramTranspose": 0.2800898551940918,
"DumpGraphAndMetadata": 0.14458084106445313,
"EliminateDivs": 0.11128425598144531,
"ExpandBatchNorm": 0.04912686347961426,
"ExpandISAMacro": 0.07448863983154297,
"FactorizeBlkDims": 0.48647499084472656,
"FactorizeThreadAxesInFreeDims": 0.05206179618835449,
"FlattenMacroLoop": 0.08013129234313965,
"GenericAccessSimplifier": 0.022362232208251953,
"InferInitValue": 1.3044743537902832,
"InferIntrinsicOnCC": 0.26564502716064453,
"InferNeuronTensor": 1.4273626804351807,
"InferNonlocalTensors": 3.070617198944092,
"InferPSumTensor": 1.8667116165161133,
"InferShardAxis": 4.046135902404785,
"InferSharedMemLoc": 0.10098028182983398,
"InlineNativeKernels": 0.046288251876831055,
"InsertCoreBarrier": 0.1175682544708252,
"InsertIOTransposes": 1.5809462070465088,
"InsertImplicitShardAxisBeforeISel": 0.37781310081481934,
"InsertLocalTransposes": 0.8543226718902588,
"InsertOffloadedTransposes": 0.0779869556427002,
"LICM": 0.10844612121582031,
"LateLegalizeInst": 0.13918232917785645,
"LateLegalizePostSplit": 0.08665323257446289,
"LateLowerReshapeOp": 0.02942204475402832,
"LateLowerTensorOp": 0.22415471076965332,
"LateNeuronInstComb": 0.9580295085906982,
"LayoutPreprocessing": 0.7032866477966309,
"LayoutPreprocessingAndAnalysis": 1.1142585277557373,
"LayoutRequirementAnalysis": 0.40787601470947266,
"LegalizeCCOpLayout": 0.052059173583984375,
"LegalizeOpLevelAlias": 0.022985219955444336,
"LegalizePartitionReduce": 0.036779165267944336,
"LegalizeSundaAccess": 0.9256985187530518,
"LegalizeSundaMacro": 0.7163646221160889,
"LegalizeType": 0.12823128700256348,
"LocalLayoutOpt": 0.5141463279724121,
"LoopFusion": 0.2369997501373291,
"LoopSplitting": 0.026223182678222656,
"LowerBroadcast": 0.04629969596862793,
"LowerCCOpBlockAxis": 0.17850804328918457,
"LowerComplexBroadcast": 0.06136465072631836,
"LowerIntrinsics": 0.898535966873169,
"LowerShardAxis": 0.19546747207641602,
"LowerTensorOp": 0.38861823081970215,
"LowerToSendRecv": 0.1479358673095703,
"LowerTranspose": 0.4128274917602539,
"MacroGeneration": 1.907278060913086,
"MaskPropagation": 0.10989689826965332,
"MemcpyElimination": 3.3127715587615967,
"MutateDataType": 0.031205177307128906,
"NeuronAliasDependencyInduction": 0.014397859573364258,
"NeuronAliasDependencyReset": 0.018595218658447266,
"NeuronInstComb": 0.28766393661499023,
"NeuronLICM": 0.24966120719909668,
"NeuronLoopFusion": 0.9968361854553223,
"NeuronLoopInterchange": 0.049851179122924805,
"NeuronSimplifier": 0.4482152462005615,
"NeuronSimplifyPredicates": 0.12775635719299316,
"NeuronValueNumbering": 0.09903788566589355,
"OptimizeAliasedCopyChain": 0.010796785354614258,
"OptimizeNKIKernels": 0.9786763191223145,
"PAGLayoutOpt": 11.095458984375,
"PComputeCutting": 0.28275370597839355,
"PGLayoutTilingPipeline": 27.67894172668457,
"PGTiling": 4.957591533660889,
"PadElimination": 0.009732246398925781,
"ParAxesAnnotation": 10.236833572387695,
"PartialLoopFusion": 1.0972201824188232,
"PartialSimdFusion": 0.529569149017334,
"PerfectLoopNest": 0.059952735900878906,
"RecognizeOpIdiom": 0.120391845703125,
"Recompute": 0.005687713623046875,
"RelaxPredicates": 0.09513998031616211,
"Rematerialization": 0.14099407196044922,
"RemoveShardedPartitionAxes": 0.6145329475402832,
"ReshapeWeights": 0.0214080810546875,
"ResolveAccessConflict": 0.1656482219696045,
"ResolveComplicatePredicates": 0.03499007225036621,
"RewriteReplicationMatmul": 0.03796839714050293,
"RewriteWeights": 0.06430387496948242,
"SFKVectorizer": 5.7498369216918945,
"ShardingPropagationAnalysis": 0.5885381698608398,
"SimpleAllReduceTiling": 0.05400824546813965,
"Simplifier": 0.07847452163696289,
"SimplifyMacroPredicates": 0.25547003746032715,
"SimplifyNeuronTensor": 0.2766604423522949,
"SimplifySlice": 0.02300572395324707,
"SimplifyTensor": 0.2370157241821289,
"SpillPSum": 0.4576537609100342,
"SplitAPUnionSets": 0.6796882152557373,
"SplitAccGrp": 0.03665971755981445,
"StaticProfiler": 0.13864398002624512,
"StaticTransposeLocalTensor": 0.2623753547668457,
"SundaISel": 1.3397884368896484,
"TCTransform": 0.025269269943237305,
"TensorInitialization": 0.16660237312316895,
"TensorOpSimplifier": 0.15259718894958496,
"TensorOpTransform": 0.8282039165496826,
"TileCCOps": 0.3066561222076416,
"TilingProfiler": 0.33243393898010254,
"TransformConvOp": 0.05726790428161621,
"TritiumFusion": 0.13151931762695313,
"ValueNumbering": 0.06717801094055176,
"VectorizeDMA": 0.5187788009643555,
"VectorizeMatMult": 0.04601097106933594,
"WeightCoalescing": 0.049105167388916016,
"ZeroSizeTensorElimination": 0.0003685951232910156
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 41972,
"StaticProfiler::AifUb": 17.11549949645996,
"StaticProfiler::ArithmeticIntensityTensorizer": 27.70398712158203,
"StaticProfiler::AverageDmaLength": 2723.7119140625,
"StaticProfiler::AverageFractalPeUtilization": 96.97119140625,
"StaticProfiler::AveragePartitionUtilization": 88.53791809082031,
"StaticProfiler::AveragePeUtilization": 81.30671691894531,
"StaticProfiler::DDRTransferBytes": 918042224,
"StaticProfiler::InternalTransferBytes": 174444896,
"StaticProfiler::LoadExpanded": 238040,
"StaticProfiler::LocalizationEfficiency": 161.8649139404297,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 167.2097930908203,
"StaticProfiler::StoreExpanded": 18609,
"StaticProfiler::TotalDMAExpanded": 256649,
"StaticProfiler::TotalDynamicInstancesCount": 57730,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50535,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 179,
"TilingProfiler::MatMultInstructionsAfterTiling": 30304,
"TilingProfiler::NumPfTransposes": 348,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 198,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 7835,
"TilingProfiler::PfTransposeInstructionsForIo": 5217,
"TilingProfiler::PfTransposeInstructionsForLocal": 508,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 2110,
"TilingProfiler::ReduceInstructionsAfterTiling": 59,
"TilingProfiler::SimdInstructionsAfterTiling": 2266,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"topk": {
"compiletime": {
"CoalesceCCOp": 0.003599405288696289,
"DMALocalityOpt": 0.003013134002685547,
"DMAProfiler": 0.0034313201904296875,
"DataStreaming": 0.006362199783325195,
"DoNothing": 0.00013327598571777344,
"ExpandISAMacro": 0.003620147705078125,
"FactorizeBlkDims": 0.010854482650756836,
"InferPSumTensor": 0.010383129119873047,
"InferSharedMemLoc": 0.0029261112213134766,
"InsertCoreBarrier": 0.003323793411254883,
"LateLegalizeInst": 0.007294178009033203,
"LateNeuronInstComb": 0.008392095565795898,
"LegalizeSundaAccess": 0.013858556747436523,
"LegalizeType": 0.009220361709594727,
"LowerBroadcast": 0.0032749176025390625,
"LowerIntrinsics": 0.0036187171936035156,
"LowerTranspose": 0.0032770633697509766,
"NeuronInstComb": 0.008351325988769531,
"NeuronLICM": 0.009110689163208008,
"NeuronSimplifyPredicates": 0.0036361217498779297,
"NeuronValueNumbering": 0.0038361549377441406,
"SFKVectorizer": 0.031067371368408203,
"SimpleAllReduceTiling": 0.0040929317474365234,
"SimplifyNeuronTensor": 0.057866573333740234,
"SpillPSum": 0.02007007598876953,
"WeightCoalescing": 0.003391265869140625
}
}
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:29d9d8ef7e5a0b735090051a93d177f1b4866fe07fc6988421cb47de73a1953f
size 3185664

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2751bdd6298e760b3ed04356f31f6ffbb7d7c37e3002f4e686a4fba8c24c2240
size 2465286

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:166776f75e13c12167868e06c05edd9ecd84b1f5005be8ff59ffb090e4b60e00
size 2443667

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:29d9d8ef7e5a0b735090051a93d177f1b4866fe07fc6988421cb47de73a1953f
size 3185664

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 4,
"bucket_n_active_tokens": false,
"buckets": [
128
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 1,
"chunked_prefill_config": null,
"context_encoding_buckets": null,
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": false,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 1,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": [
128
],
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3d7d771432f2a5b315c36747141a382cd677d56bd52622fbc82564ea1e8d3890
size 3380710

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_f53407701fa4882a24c0+55c11e15.hlo_module.pb --output model.MODULE_f53407701fa4882a24c0+55c11e15.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk1/log-neuron-cc.txt"]

View File

@@ -0,0 +1,590 @@
{
"Average": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 97.62759399414063,
"StaticProfiler::AveragePartitionUtilization": 90.06017303466797,
"StaticProfiler::AveragePeUtilization": 82.14405822753906,
"StaticProfiler::LocalizationEfficiency": 160.6605224609375,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 165.92486572265625,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
}
},
"Count": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 1,
"StaticProfiler::AveragePartitionUtilization": 1,
"StaticProfiler::AveragePeUtilization": 1,
"StaticProfiler::LocalizationEfficiency": 1,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
}
},
"Sum": {
"compiletime": {
"AGOrderingAnalysisPass": 5.160665512084961,
"AffinePredicateResolution": 0.7100803852081299,
"AliasDependencyElimination": 0.0032160282135009766,
"AliasDependencyInduction": 17.29242706298828,
"AliasDependencyReset": 18.28659439086914,
"BFComputeCutting": 0.24476861953735352,
"BirCodeGenLoop": 6.947810173034668,
"CCOpFusion": 0.6929187774658203,
"CanonicalizeConv": 2.9000000722589903e-05,
"CanonicalizeDAGForPGTiling": 0.21501421928405762,
"CanonicalizeForTensorizer": 0.0008399999933317304,
"CanonicalizeIR": 0.8826379776000977,
"Canonicalizer": 0.014088000170886517,
"CoalesceCCOp": 0.5657715797424316,
"CommuteConcat": 0.028485536575317383,
"DMALocalityOpt": 0.042168617248535156,
"DMAProfiler": 0.1519615650177002,
"DMATilingProfiler": 0.09857678413391113,
"DataLocalityOpt": 2.827857255935669,
"DataStreaming": 0.20627403259277344,
"DeConcat": 0.08719110488891602,
"DeadCodeElimination": 0.03139615058898926,
"DeadStoreElimination": 1.1847929954528809,
"DelinearIndices": 0.4130244255065918,
"Delinearization": 0.35245656967163086,
"DelinearizeSPMD": 0.4181056022644043,
"DoNothing": 0.00046539306640625,
"DramToDramTranspose": 0.30753135681152344,
"DumpGraphAndMetadata": 0.2011098861694336,
"EliminateDivs": 1.3804805278778076,
"ExpandBatchNorm": 1.1923854351043701,
"ExpandISAMacro": 0.08926725387573242,
"FactorizeBlkDims": 0.5252444744110107,
"FactorizeThreadAxesInFreeDims": 0.1739037036895752,
"FlattenMacroLoop": 0.07839345932006836,
"GenericAccessSimplifier": 0.025495052337646484,
"HoistCompute": 6.199999916134402e-05,
"IdentifyCrossPassTensors": 0.0005760000203736126,
"InferInitValue": 1.3857665061950684,
"InferIntrinsicOnCC": 0.646845817565918,
"InferNeuronTensor": 2.0493955612182617,
"InferNonlocalTensors": 6.630102634429932,
"InferPSumTensor": 1.2878854274749756,
"InferShardAxis": 10.588101387023926,
"InferSharedMemLoc": 0.10591363906860352,
"InlineNativeKernels": 0.0488896369934082,
"InsertCoreBarrier": 0.3818776607513428,
"InsertIOTransposes": 0.8546113967895508,
"InsertImplicitShardAxisBeforeISel": 0.36914730072021484,
"InsertLocalTransposes": 1.5642907619476318,
"InsertOffloadedTransposes": 0.12299966812133789,
"LICM": 0.12427496910095215,
"LateLegalizeInst": 0.4006483554840088,
"LateLegalizePostSplit": 0.09814929962158203,
"LateLowerReshapeOp": 0.0431976318359375,
"LateLowerTensorOp": 5.883858680725098,
"LateNeuronInstComb": 1.04233980178833,
"LayoutPreprocessing": 1.2663531303405762,
"LayoutPreprocessingAndAnalysis": 1.9893858432769775,
"LayoutRequirementAnalysis": 0.6988728046417236,
"LegalizeCCOpLayout": 1.499981164932251,
"LegalizeOpLevelAlias": 0.6842360496520996,
"LegalizePartitionReduce": 0.09090352058410645,
"LegalizeSundaAccess": 3.133746862411499,
"LegalizeSundaMacro": 0.6402661800384521,
"LegalizeType": 0.1407451629638672,
"LocalLayoutOpt": 0.7939743995666504,
"LoopFusion": 0.29726386070251465,
"LoopSplitting": 0.08333015441894531,
"LowerBroadcast": 0.06013798713684082,
"LowerCCOpBlockAxis": 1.4568629264831543,
"LowerComplexBroadcast": 0.06818914413452148,
"LowerIntrinsics": 0.9435675144195557,
"LowerShardAxis": 0.22619009017944336,
"LowerTensorOp": 3.0802407264709473,
"LowerToSendRecv": 0.19478487968444824,
"LowerTranspose": 0.5133018493652344,
"MacroGeneration": 4.587047100067139,
"MaskPropagation": 0.1672959327697754,
"MemcastMotion": 0.00015999999595806003,
"MemcpyElimination": 29.92277717590332,
"MutateDataType": 0.04070854187011719,
"NeuronAliasDependencyInduction": 0.018942594528198242,
"NeuronAliasDependencyReset": 0.02425408363342285,
"NeuronInstComb": 0.4308168888092041,
"NeuronLICM": 0.29584765434265137,
"NeuronLoopFusion": 1.117248296737671,
"NeuronLoopInterchange": 0.0636141300201416,
"NeuronSimplifier": 0.704033374786377,
"NeuronSimplifyPredicates": 0.2319803237915039,
"NeuronValueNumbering": 0.11443066596984863,
"OptimizeAliasedCopyChain": 0.2889397144317627,
"OptimizeNKIKernels": 1.2374975681304932,
"PAGLayoutOpt": 18.6531982421875,
"PComputeCutting": 0.5685915946960449,
"PGLayoutTilingPipeline": 53.59619903564453,
"PGTiling": 11.116703033447266,
"PadElimination": 0.012248039245605469,
"ParAxesAnnotation": 17.06294822692871,
"PartialLoopFusion": 1.3277881145477295,
"PartialSimdFusion": 0.7600483894348145,
"PenguinizeFunctions": 0.001062000053934753,
"PerfectLoopNest": 0.061771392822265625,
"PruneFunctions": 0.0004830000107176602,
"RecognizeOpIdiom": 0.12406373023986816,
"Recompute": 0.00886678695678711,
"RelaxPredicates": 0.11265873908996582,
"Rematerialization": 0.18548583984375,
"RemoveOptimizationBarriers": 0.009134000167250633,
"RemoveShardedPartitionAxes": 1.4166390895843506,
"ReshapeWeights": 0.02198624610900879,
"ResolveAccessConflict": 0.20142865180969238,
"ResolveComplicatePredicates": 0.6187732219696045,
"RewriteReplicationMatmul": 0.04161477088928223,
"RewriteWeights": 0.0676727294921875,
"SFKVectorizer": 11.220393180847168,
"ScatterMotion": 0.003527000080794096,
"ShardingPropagationAnalysis": 0.8252537250518799,
"SimpleAllReduceTiling": 0.1859297752380371,
"Simplifier": 0.09509730339050293,
"SimplifyMacroPredicates": 0.26697540283203125,
"SimplifyNeuronTensor": 0.41252875328063965,
"SimplifySlice": 0.02629375457763672,
"SimplifyTensor": 0.42005443572998047,
"SpillPSum": 0.5884366035461426,
"SplitAPUnionSets": 0.4571385383605957,
"SplitAccGrp": 0.052629709243774414,
"StaticProfiler": 0.12954020500183105,
"StaticTransposeLocalTensor": 0.36740827560424805,
"SundaISel": 1.4737660884857178,
"TCTransform": 0.031054019927978516,
"TensorInitialization": 0.18288874626159668,
"TensorOpSimplifier": 3.2880165576934814,
"TensorOpTransform": 18.079126358032227,
"TensorizerLegalizationPass": 0.0004990000161342323,
"TileCCOps": 0.17596793174743652,
"TilingProfiler": 0.3990769386291504,
"TransformConvOp": 1.2077322006225586,
"TritiumFusion": 0.2976958751678467,
"ValueNumbering": 0.10138130187988281,
"VectorizeDMA": 0.8864037990570068,
"VectorizeMatMult": 0.05278921127319336,
"VerifySupportedOps": 0.0004729999927803874,
"WeightCoalescing": 0.08338785171508789,
"ZeroSizeTensorElimination": 0.0006525516510009766,
"algsimp": 0.0014349999837577343,
"batchnorm_expander": 0.000818000000435859,
"boundary-marker-removal": 0.0003020000003743917,
"call-inliner": 0.00018099999579135329,
"canonicalize-boundary-marker": 0.0005639999872073531,
"collective-stream-id-checker": 0.002294000005349517,
"comparison-expander": 0.0012120000319555402,
"computation-deduplicator": 0.0011109999613836408,
"config-lowering": 0.00042299999040551484,
"constant_folding": 0.0001289999927394092,
"cse": 0.0007570000016130507,
"dce": 7.300000288523734e-05,
"dynamic-slice-transpose": 0.0002640000020619482,
"eliminate-redundant-compare": 0.0001250000059371814,
"emit-offloaded-dropout": 0.0006559999892488122,
"flatten-call-graph": 0.0003330000035930425,
"fuse-send-recv": 0.013015000149607658,
"hilo-conditional-to-select": 0.00016799999866634607,
"hilo::LegalizeAlias": 0.0036430000327527523,
"hilo::NeuronInstCombine": 1.1000000085914508e-05,
"hilo::NeuronOpFusion": 0.00011600000289035961,
"hilo::ReplaceTokenTypeWithU8Pass": 0.0003260000084992498,
"hilo::ScheduleFusion": 3.300000025774352e-05,
"hilo::SixtyFourHack": 0.0013909999979659915,
"hilo::VerifyAliasing": 0.00014600000577047467,
"hlo-mac-count": 0.019317999482154846,
"io-con-pipe-begin": 0.0002589999930933118,
"io-con-pipe-end": 9.999999974752427e-07,
"io-layout-normalization": 0.07799100130796432,
"legalize-ccops-for-tensorizer": 5.2999999752501026e-05,
"legalize-compare": 0.0006939999875612557,
"lower-argminmax-custom-call": 0.0002629999944474548,
"map-inline": 0.06807900220155716,
"metadata-naming": 0.07221200317144394,
"mlir::detail::OpToOpPassAdaptor": 0.007284999825060368,
"mlir::hlo::MhloToPyPenguin": 0.6235949993133545,
"mlir::mhlo::LowerComplexExtraPass": 0.00916799996048212,
"mlir::mhlo::LowerComplexPass": 0.0006799999973736703,
"native-to-custom-softmax": 0.0017719999887049198,
"native-to-custom-softmax-dx": 0.0017500000540167093,
"neuron-hlo-verifier": 0.2390509992837906,
"operand_upcaster": 0.070872001349926,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 9.999999747378752e-06,
"post-partition-simplification": 0.21146699786186218,
"pre-hlo-begin": 8.600000001024455e-05,
"pre-hlo-end": 9.999999974752427e-07,
"replace-minimum-constant": 0.00019700000120792538,
"reshape-mover": 5.400000009103678e-05,
"simplify-concat": 0.002297000028192997,
"simplify-while-loops": 4.600000102072954e-05,
"transform-variadic-reduce": 0.00037900000461377203,
"tuple-simplifier": 0.0001340000017080456,
"unpack-nested-aws-ntwsr": 0.00025400001322850585,
"unroll-while-loop": 7.000000096013537e-06
},
"hilo": {
"HloMacCount": 1766645760.0,
"Traffic": 1252618368.0
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 42842,
"StaticProfiler::AifUb": 17.765029907226563,
"StaticProfiler::ArithmeticIntensityTensorizer": 28.541391372680664,
"StaticProfiler::AverageDmaLength": 3512.794677734375,
"StaticProfiler::DDRTransferBytes": 924925552,
"StaticProfiler::InternalTransferBytes": 182108768,
"StaticProfiler::LoadExpanded": 181148,
"StaticProfiler::StoreExpanded": 4273,
"StaticProfiler::TotalDMAExpanded": 185421,
"StaticProfiler::TotalDynamicInstancesCount": 57693,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50217,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 123,
"TilingProfiler::MatMultInstructionsAfterTiling": 30976,
"TilingProfiler::NumPfTransposes": 348,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 198,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 7892,
"TilingProfiler::PfTransposeInstructionsForIo": 5666,
"TilingProfiler::PfTransposeInstructionsForLocal": 564,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
"TilingProfiler::ReduceInstructionsAfterTiling": 115,
"TilingProfiler::SimdInstructionsAfterTiling": 2295,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"all": {
"compiletime": {
"CanonicalizeConv": 2.9000000722589903e-05,
"CanonicalizeForTensorizer": 0.0008399999933317304,
"Canonicalizer": 0.014088000170886517,
"HoistCompute": 6.199999916134402e-05,
"IdentifyCrossPassTensors": 0.0005760000203736126,
"MemcastMotion": 0.00015999999595806003,
"PenguinizeFunctions": 0.001062000053934753,
"PruneFunctions": 0.0004830000107176602,
"RemoveOptimizationBarriers": 0.009134000167250633,
"ScatterMotion": 0.003527000080794096,
"TensorizerLegalizationPass": 0.0004990000161342323,
"VerifySupportedOps": 0.0004729999927803874,
"algsimp": 0.0014349999837577343,
"batchnorm_expander": 0.000818000000435859,
"boundary-marker-removal": 0.0003020000003743917,
"call-inliner": 0.00018099999579135329,
"canonicalize-boundary-marker": 0.0005639999872073531,
"collective-stream-id-checker": 0.002294000005349517,
"comparison-expander": 0.0012120000319555402,
"computation-deduplicator": 0.0011109999613836408,
"config-lowering": 0.00042299999040551484,
"constant_folding": 0.0001289999927394092,
"cse": 0.0007570000016130507,
"dce": 7.300000288523734e-05,
"dynamic-slice-transpose": 0.0002640000020619482,
"eliminate-redundant-compare": 0.0001250000059371814,
"emit-offloaded-dropout": 0.0006559999892488122,
"flatten-call-graph": 0.0003330000035930425,
"fuse-send-recv": 0.013015000149607658,
"hilo-conditional-to-select": 0.00016799999866634607,
"hilo::LegalizeAlias": 0.0036430000327527523,
"hilo::NeuronInstCombine": 1.1000000085914508e-05,
"hilo::NeuronOpFusion": 0.00011600000289035961,
"hilo::ReplaceTokenTypeWithU8Pass": 0.0003260000084992498,
"hilo::ScheduleFusion": 3.300000025774352e-05,
"hilo::SixtyFourHack": 0.0013909999979659915,
"hilo::VerifyAliasing": 0.00014600000577047467,
"hlo-mac-count": 0.019317999482154846,
"io-con-pipe-begin": 0.0002589999930933118,
"io-con-pipe-end": 9.999999974752427e-07,
"io-layout-normalization": 0.07799100130796432,
"legalize-ccops-for-tensorizer": 5.2999999752501026e-05,
"legalize-compare": 0.0006939999875612557,
"lower-argminmax-custom-call": 0.0002629999944474548,
"map-inline": 0.06807900220155716,
"metadata-naming": 0.07221200317144394,
"mlir::detail::OpToOpPassAdaptor": 0.007284999825060368,
"mlir::hlo::MhloToPyPenguin": 0.6235949993133545,
"mlir::mhlo::LowerComplexExtraPass": 0.00916799996048212,
"mlir::mhlo::LowerComplexPass": 0.0006799999973736703,
"native-to-custom-softmax": 0.0017719999887049198,
"native-to-custom-softmax-dx": 0.0017500000540167093,
"neuron-hlo-verifier": 0.2390509992837906,
"operand_upcaster": 0.070872001349926,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 9.999999747378752e-06,
"post-partition-simplification": 0.21146699786186218,
"pre-hlo-begin": 8.600000001024455e-05,
"pre-hlo-end": 9.999999974752427e-07,
"replace-minimum-constant": 0.00019700000120792538,
"reshape-mover": 5.400000009103678e-05,
"simplify-concat": 0.002297000028192997,
"simplify-while-loops": 4.600000102072954e-05,
"transform-variadic-reduce": 0.00037900000461377203,
"tuple-simplifier": 0.0001340000017080456,
"unpack-nested-aws-ntwsr": 0.00025400001322850585,
"unroll-while-loop": 7.000000096013537e-06
}
},
"cumsum": {
"compiletime": {
"CoalesceCCOp": 0.0002148151397705078,
"DMALocalityOpt": 0.0001659393310546875,
"DMAProfiler": 0.0008566379547119141,
"DataStreaming": 0.00026869773864746094,
"DoNothing": 0.00015687942504882813,
"ExpandISAMacro": 0.0004825592041015625,
"FactorizeBlkDims": 0.0004360675811767578,
"InferPSumTensor": 0.0005650520324707031,
"InferSharedMemLoc": 0.0002880096435546875,
"InsertCoreBarrier": 0.0002799034118652344,
"LateLegalizeInst": 0.0003802776336669922,
"LateNeuronInstComb": 0.0006039142608642578,
"LegalizeSundaAccess": 0.001421213150024414,
"LegalizeType": 0.0002644062042236328,
"LowerBroadcast": 0.0002384185791015625,
"LowerIntrinsics": 0.0002446174621582031,
"LowerTranspose": 0.00022077560424804688,
"NeuronInstComb": 0.0005753040313720703,
"NeuronLICM": 0.00037217140197753906,
"NeuronSimplifyPredicates": 0.0022978782653808594,
"NeuronValueNumbering": 0.0004074573516845703,
"SFKVectorizer": 0.002466440200805664,
"SimpleAllReduceTiling": 0.0002262592315673828,
"SimplifyNeuronTensor": 0.0005323886871337891,
"SpillPSum": 0.00045490264892578125,
"WeightCoalescing": 0.0002465248107910156
}
},
"sg00": {
"hilo": {
"ArithmeticIntensity": 2.8207247257232666,
"HloMacCount": 1766645760.0,
"Traffic": 1252618368.0
}
},
"sg0000": {
"compiletime": {
"AGOrderingAnalysisPass": 5.160665512084961,
"AffinePredicateResolution": 0.7100803852081299,
"AliasDependencyElimination": 0.0032160282135009766,
"AliasDependencyInduction": 17.29242706298828,
"AliasDependencyReset": 18.28659439086914,
"BFComputeCutting": 0.24476861953735352,
"BirCodeGenLoop": 6.947810173034668,
"CCOpFusion": 0.6929187774658203,
"CanonicalizeDAGForPGTiling": 0.21501421928405762,
"CanonicalizeIR": 0.8826379776000977,
"CoalesceCCOp": 0.5618836879730225,
"CommuteConcat": 0.028485536575317383,
"DMALocalityOpt": 0.03933858871459961,
"DMAProfiler": 0.14742827415466309,
"DMATilingProfiler": 0.09857678413391113,
"DataLocalityOpt": 2.827857255935669,
"DataStreaming": 0.19969892501831055,
"DeConcat": 0.08719110488891602,
"DeadCodeElimination": 0.03139615058898926,
"DeadStoreElimination": 1.1847929954528809,
"DelinearIndices": 0.4130244255065918,
"Delinearization": 0.35245656967163086,
"DelinearizeSPMD": 0.4181056022644043,
"DoNothing": 0.0001385211944580078,
"DramToDramTranspose": 0.30753135681152344,
"DumpGraphAndMetadata": 0.2011098861694336,
"EliminateDivs": 1.3804805278778076,
"ExpandBatchNorm": 1.1923854351043701,
"ExpandISAMacro": 0.08506417274475098,
"FactorizeBlkDims": 0.5093107223510742,
"FactorizeThreadAxesInFreeDims": 0.1739037036895752,
"FlattenMacroLoop": 0.07839345932006836,
"GenericAccessSimplifier": 0.025495052337646484,
"InferInitValue": 1.3857665061950684,
"InferIntrinsicOnCC": 0.646845817565918,
"InferNeuronTensor": 2.0493955612182617,
"InferNonlocalTensors": 6.630102634429932,
"InferPSumTensor": 1.275925636291504,
"InferShardAxis": 10.588101387023926,
"InferSharedMemLoc": 0.102783203125,
"InlineNativeKernels": 0.0488896369934082,
"InsertCoreBarrier": 0.3783996105194092,
"InsertIOTransposes": 0.8546113967895508,
"InsertImplicitShardAxisBeforeISel": 0.36914730072021484,
"InsertLocalTransposes": 1.5642907619476318,
"InsertOffloadedTransposes": 0.12299966812133789,
"LICM": 0.12427496910095215,
"LateLegalizeInst": 0.3925168514251709,
"LateLegalizePostSplit": 0.09814929962158203,
"LateLowerReshapeOp": 0.0431976318359375,
"LateLowerTensorOp": 5.883858680725098,
"LateNeuronInstComb": 1.0332238674163818,
"LayoutPreprocessing": 1.2663531303405762,
"LayoutPreprocessingAndAnalysis": 1.9893858432769775,
"LayoutRequirementAnalysis": 0.6988728046417236,
"LegalizeCCOpLayout": 1.499981164932251,
"LegalizeOpLevelAlias": 0.6842360496520996,
"LegalizePartitionReduce": 0.09090352058410645,
"LegalizeSundaAccess": 3.1170387268066406,
"LegalizeSundaMacro": 0.6402661800384521,
"LegalizeType": 0.1309196949005127,
"LocalLayoutOpt": 0.7939743995666504,
"LoopFusion": 0.29726386070251465,
"LoopSplitting": 0.08333015441894531,
"LowerBroadcast": 0.05664563179016113,
"LowerCCOpBlockAxis": 1.4568629264831543,
"LowerComplexBroadcast": 0.06818914413452148,
"LowerIntrinsics": 0.9396772384643555,
"LowerShardAxis": 0.22619009017944336,
"LowerTensorOp": 3.0802407264709473,
"LowerToSendRecv": 0.19478487968444824,
"LowerTranspose": 0.5097734928131104,
"MacroGeneration": 4.587047100067139,
"MaskPropagation": 0.1672959327697754,
"MemcpyElimination": 29.92277717590332,
"MutateDataType": 0.04070854187011719,
"NeuronAliasDependencyInduction": 0.018942594528198242,
"NeuronAliasDependencyReset": 0.02425408363342285,
"NeuronInstComb": 0.42151784896850586,
"NeuronLICM": 0.2860753536224365,
"NeuronLoopFusion": 1.117248296737671,
"NeuronLoopInterchange": 0.0636141300201416,
"NeuronSimplifier": 0.704033374786377,
"NeuronSimplifyPredicates": 0.22606325149536133,
"NeuronValueNumbering": 0.11009359359741211,
"OptimizeAliasedCopyChain": 0.2889397144317627,
"OptimizeNKIKernels": 1.2374975681304932,
"PAGLayoutOpt": 18.6531982421875,
"PComputeCutting": 0.5685915946960449,
"PGLayoutTilingPipeline": 53.59619903564453,
"PGTiling": 11.116703033447266,
"PadElimination": 0.012248039245605469,
"ParAxesAnnotation": 17.06294822692871,
"PartialLoopFusion": 1.3277881145477295,
"PartialSimdFusion": 0.7600483894348145,
"PerfectLoopNest": 0.061771392822265625,
"RecognizeOpIdiom": 0.12406373023986816,
"Recompute": 0.00886678695678711,
"RelaxPredicates": 0.11265873908996582,
"Rematerialization": 0.18548583984375,
"RemoveShardedPartitionAxes": 1.4166390895843506,
"ReshapeWeights": 0.02198624610900879,
"ResolveAccessConflict": 0.20142865180969238,
"ResolveComplicatePredicates": 0.6187732219696045,
"RewriteReplicationMatmul": 0.04161477088928223,
"RewriteWeights": 0.0676727294921875,
"SFKVectorizer": 11.18409252166748,
"ShardingPropagationAnalysis": 0.8252537250518799,
"SimpleAllReduceTiling": 0.18207907676696777,
"Simplifier": 0.09509730339050293,
"SimplifyMacroPredicates": 0.26697540283203125,
"SimplifyNeuronTensor": 0.3594348430633545,
"SimplifySlice": 0.02629375457763672,
"SimplifyTensor": 0.42005443572998047,
"SpillPSum": 0.5635168552398682,
"SplitAPUnionSets": 0.4571385383605957,
"SplitAccGrp": 0.052629709243774414,
"StaticProfiler": 0.12954020500183105,
"StaticTransposeLocalTensor": 0.36740827560424805,
"SundaISel": 1.4737660884857178,
"TCTransform": 0.031054019927978516,
"TensorInitialization": 0.18288874626159668,
"TensorOpSimplifier": 3.2880165576934814,
"TensorOpTransform": 18.079126358032227,
"TileCCOps": 0.17596793174743652,
"TilingProfiler": 0.3990769386291504,
"TransformConvOp": 1.2077322006225586,
"TritiumFusion": 0.2976958751678467,
"ValueNumbering": 0.10138130187988281,
"VectorizeDMA": 0.8864037990570068,
"VectorizeMatMult": 0.05278921127319336,
"WeightCoalescing": 0.07968640327453613,
"ZeroSizeTensorElimination": 0.0006525516510009766
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 42842,
"StaticProfiler::AifUb": 17.765029907226563,
"StaticProfiler::ArithmeticIntensityTensorizer": 28.541391372680664,
"StaticProfiler::AverageDmaLength": 3512.794677734375,
"StaticProfiler::AverageFractalPeUtilization": 97.62759399414063,
"StaticProfiler::AveragePartitionUtilization": 90.06017303466797,
"StaticProfiler::AveragePeUtilization": 82.14405822753906,
"StaticProfiler::DDRTransferBytes": 924925552,
"StaticProfiler::InternalTransferBytes": 182108768,
"StaticProfiler::LoadExpanded": 181148,
"StaticProfiler::LocalizationEfficiency": 160.6605224609375,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 165.92486572265625,
"StaticProfiler::StoreExpanded": 4273,
"StaticProfiler::TotalDMAExpanded": 185421,
"StaticProfiler::TotalDynamicInstancesCount": 57693,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 50217,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 123,
"TilingProfiler::MatMultInstructionsAfterTiling": 30976,
"TilingProfiler::NumPfTransposes": 348,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 198,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 7892,
"TilingProfiler::PfTransposeInstructionsForIo": 5666,
"TilingProfiler::PfTransposeInstructionsForLocal": 564,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
"TilingProfiler::ReduceInstructionsAfterTiling": 115,
"TilingProfiler::SimdInstructionsAfterTiling": 2295,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"topk": {
"compiletime": {
"CoalesceCCOp": 0.003673076629638672,
"DMALocalityOpt": 0.0026640892028808594,
"DMAProfiler": 0.0036766529083251953,
"DataStreaming": 0.00630640983581543,
"DoNothing": 0.00016999244689941406,
"ExpandISAMacro": 0.003720521926879883,
"FactorizeBlkDims": 0.015497684478759766,
"InferPSumTensor": 0.011394739151000977,
"InferSharedMemLoc": 0.002842426300048828,
"InsertCoreBarrier": 0.0031981468200683594,
"LateLegalizeInst": 0.0077512264251708984,
"LateNeuronInstComb": 0.008512020111083984,
"LegalizeSundaAccess": 0.015286922454833984,
"LegalizeType": 0.00956106185913086,
"LowerBroadcast": 0.003253936767578125,
"LowerIntrinsics": 0.003645658493041992,
"LowerTranspose": 0.0033075809478759766,
"NeuronInstComb": 0.008723735809326172,
"NeuronLICM": 0.009400129318237305,
"NeuronSimplifyPredicates": 0.0036191940307617188,
"NeuronValueNumbering": 0.003929615020751953,
"SFKVectorizer": 0.033834218978881836,
"SimpleAllReduceTiling": 0.003624439239501953,
"SimplifyNeuronTensor": 0.05256152153015137,
"SpillPSum": 0.024464845657348633,
"WeightCoalescing": 0.003454923629760742
}
}
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cda6f9bc52718ce93e1fb58a7aa253d606cf545bda03b396e531716e6a9ce678
size 3154944

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:fd8b7f01387ce1b80997167a938d7a05e712ade8a8271b5da3b01c446894cbd8
size 2464555

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d9364e7ab828d2cbc2c983e088e9c293cdaddc1137d948ee2d3c625e9eab02f7
size 2550451

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cda6f9bc52718ce93e1fb58a7aa253d606cf545bda03b396e531716e6a9ce678
size 3154944

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 4,
"bucket_n_active_tokens": false,
"buckets": [
256
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 1,
"chunked_prefill_config": null,
"context_encoding_buckets": null,
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": false,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 1,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": [
256
],
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_c60e20a111715e620aa2+6bb8acc9.hlo_module.pb --output model.MODULE_c60e20a111715e620aa2+6bb8acc9.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk2/log-neuron-cc.txt"]

View File

@@ -0,0 +1,590 @@
{
"Average": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 97.78141784667969,
"StaticProfiler::AveragePartitionUtilization": 90.33787536621094,
"StaticProfiler::AveragePeUtilization": 81.002685546875,
"StaticProfiler::LocalizationEfficiency": 155.71730041503906,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 160.65768432617188,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
}
},
"Count": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 1,
"StaticProfiler::AveragePartitionUtilization": 1,
"StaticProfiler::AveragePeUtilization": 1,
"StaticProfiler::LocalizationEfficiency": 1,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
}
},
"Sum": {
"compiletime": {
"AGOrderingAnalysisPass": 6.775559425354004,
"AffinePredicateResolution": 0.7124242782592773,
"AliasDependencyElimination": 0.0037508010864257813,
"AliasDependencyInduction": 17.88080596923828,
"AliasDependencyReset": 19.01345443725586,
"BFComputeCutting": 0.14321279525756836,
"BirCodeGenLoop": 4.049878120422363,
"CCOpFusion": 0.9675219058990479,
"CanonicalizeConv": 0.0,
"CanonicalizeDAGForPGTiling": 0.18202924728393555,
"CanonicalizeForTensorizer": 0.0005849999724887311,
"CanonicalizeIR": 0.8085732460021973,
"Canonicalizer": 0.015991000458598137,
"CoalesceCCOp": 0.25977158546447754,
"CommuteConcat": 0.027615070343017578,
"DMALocalityOpt": 0.048050880432128906,
"DMAProfiler": 0.09339690208435059,
"DMATilingProfiler": 0.09855914115905762,
"DataLocalityOpt": 2.8309335708618164,
"DataStreaming": 0.2996983528137207,
"DeConcat": 0.08213448524475098,
"DeadCodeElimination": 0.030428647994995117,
"DeadStoreElimination": 1.051145315170288,
"DelinearIndices": 0.5529322624206543,
"Delinearization": 0.14493751525878906,
"DelinearizeSPMD": 0.181105375289917,
"DoNothing": 0.0003535747528076172,
"DramToDramTranspose": 0.5176758766174316,
"DumpGraphAndMetadata": 0.1749565601348877,
"EliminateDivs": 1.6087908744812012,
"ExpandBatchNorm": 1.096595048904419,
"ExpandISAMacro": 0.09282922744750977,
"FactorizeBlkDims": 0.5121583938598633,
"FactorizeThreadAxesInFreeDims": 0.08890652656555176,
"FlattenMacroLoop": 0.08152222633361816,
"GenericAccessSimplifier": 0.02545022964477539,
"HoistCompute": 9.899999713525176e-05,
"IdentifyCrossPassTensors": 0.00011500000255182385,
"InferInitValue": 1.2978432178497314,
"InferIntrinsicOnCC": 0.7616860866546631,
"InferNeuronTensor": 2.6053829193115234,
"InferNonlocalTensors": 7.113053321838379,
"InferPSumTensor": 4.450908660888672,
"InferShardAxis": 12.849608421325684,
"InferSharedMemLoc": 0.19225406646728516,
"InlineNativeKernels": 0.08442234992980957,
"InsertCoreBarrier": 0.15249347686767578,
"InsertIOTransposes": 0.8695223331451416,
"InsertImplicitShardAxisBeforeISel": 0.49620485305786133,
"InsertLocalTransposes": 1.934779167175293,
"InsertOffloadedTransposes": 0.23423099517822266,
"LICM": 0.1213834285736084,
"LateLegalizeInst": 0.28924560546875,
"LateLegalizePostSplit": 0.22411513328552246,
"LateLowerReshapeOp": 0.033789873123168945,
"LateLowerTensorOp": 4.297260761260986,
"LateNeuronInstComb": 1.0230679512023926,
"LayoutPreprocessing": 1.8572075366973877,
"LayoutPreprocessingAndAnalysis": 2.2904410362243652,
"LayoutRequirementAnalysis": 0.4237644672393799,
"LegalizeCCOpLayout": 1.1935856342315674,
"LegalizeOpLevelAlias": 0.7793729305267334,
"LegalizePartitionReduce": 0.09432244300842285,
"LegalizeSundaAccess": 1.0004379749298096,
"LegalizeSundaMacro": 0.7198967933654785,
"LegalizeType": 0.1766369342803955,
"LocalLayoutOpt": 0.6666848659515381,
"LoopFusion": 0.29988932609558105,
"LoopSplitting": 0.05208778381347656,
"LowerBroadcast": 0.08215570449829102,
"LowerCCOpBlockAxis": 0.2862672805786133,
"LowerComplexBroadcast": 0.09356045722961426,
"LowerIntrinsics": 1.3606979846954346,
"LowerShardAxis": 0.39397311210632324,
"LowerTensorOp": 4.018380165100098,
"LowerToSendRecv": 0.27350616455078125,
"LowerTranspose": 0.5171041488647461,
"MacroGeneration": 2.0922586917877197,
"MaskPropagation": 0.12857651710510254,
"MemcastMotion": 0.0002789999998640269,
"MemcpyElimination": 30.280338287353516,
"MutateDataType": 0.03657245635986328,
"NeuronAliasDependencyInduction": 0.026292085647583008,
"NeuronAliasDependencyReset": 0.03253936767578125,
"NeuronInstComb": 0.36768054962158203,
"NeuronLICM": 0.3019218444824219,
"NeuronLoopFusion": 1.711080551147461,
"NeuronLoopInterchange": 0.08134770393371582,
"NeuronSimplifier": 0.5104148387908936,
"NeuronSimplifyPredicates": 0.24121522903442383,
"NeuronValueNumbering": 0.11496901512145996,
"OptimizeAliasedCopyChain": 0.40732264518737793,
"OptimizeNKIKernels": 1.0493371486663818,
"PAGLayoutOpt": 18.23680305480957,
"PComputeCutting": 0.5901892185211182,
"PGLayoutTilingPipeline": 54.30891036987305,
"PGTiling": 10.472368240356445,
"PadElimination": 0.0187380313873291,
"ParAxesAnnotation": 16.293506622314453,
"PartialLoopFusion": 1.5408625602722168,
"PartialSimdFusion": 0.8349573612213135,
"PenguinizeFunctions": 0.0005789999850094318,
"PerfectLoopNest": 0.07371997833251953,
"PruneFunctions": 0.00018899999849963933,
"RecognizeOpIdiom": 0.12681221961975098,
"Recompute": 0.007548809051513672,
"RelaxPredicates": 0.11768078804016113,
"Rematerialization": 0.15080475807189941,
"RemoveOptimizationBarriers": 0.0004290000069886446,
"RemoveShardedPartitionAxes": 1.1734652519226074,
"ReshapeWeights": 0.022508859634399414,
"ResolveAccessConflict": 0.20145153999328613,
"ResolveComplicatePredicates": 0.3191838264465332,
"RewriteReplicationMatmul": 0.039293527603149414,
"RewriteWeights": 0.0773766040802002,
"SFKVectorizer": 11.08578872680664,
"ScatterMotion": 0.005127000156790018,
"ShardingPropagationAnalysis": 0.665086030960083,
"SimpleAllReduceTiling": 0.07452917098999023,
"Simplifier": 0.09160900115966797,
"SimplifyMacroPredicates": 0.28479433059692383,
"SimplifyNeuronTensor": 0.5389096736907959,
"SimplifySlice": 0.02684783935546875,
"SimplifyTensor": 0.2644212245941162,
"SpillPSum": 1.0592920780181885,
"SplitAPUnionSets": 1.0491511821746826,
"SplitAccGrp": 0.055588722229003906,
"StaticProfiler": 0.39172840118408203,
"StaticTransposeLocalTensor": 0.7211995124816895,
"SundaISel": 1.5357580184936523,
"TCTransform": 0.029030561447143555,
"TensorInitialization": 0.20336008071899414,
"TensorOpSimplifier": 3.100177049636841,
"TensorOpTransform": 20.175758361816406,
"TensorizerLegalizationPass": 0.000455000001238659,
"TileCCOps": 0.17463970184326172,
"TilingProfiler": 0.7014575004577637,
"TransformConvOp": 0.9923229217529297,
"TritiumFusion": 0.33277034759521484,
"ValueNumbering": 0.08796954154968262,
"VectorizeDMA": 0.520815372467041,
"VectorizeMatMult": 0.08555078506469727,
"VerifySupportedOps": 0.0002469999890308827,
"WeightCoalescing": 0.09020423889160156,
"ZeroSizeTensorElimination": 0.0006709098815917969,
"algsimp": 0.0018210000125691295,
"batchnorm_expander": 0.0022700000554323196,
"boundary-marker-removal": 0.0009800000116229057,
"call-inliner": 0.0002410000015515834,
"canonicalize-boundary-marker": 0.0009430000209249556,
"collective-stream-id-checker": 0.0008960000122897327,
"comparison-expander": 0.000623999978415668,
"computation-deduplicator": 0.003625999903306365,
"config-lowering": 0.00017899999511428177,
"constant_folding": 0.0668639987707138,
"cse": 0.001184999942779541,
"dce": 6.399999983841553e-05,
"dynamic-slice-transpose": 0.0003530000103637576,
"eliminate-redundant-compare": 0.00014400000509340316,
"emit-offloaded-dropout": 0.0007820000173524022,
"flatten-call-graph": 0.0009519999730400741,
"fuse-send-recv": 0.07950299978256226,
"hilo-conditional-to-select": 0.00015700000221841037,
"hilo::LegalizeAlias": 0.003444999922066927,
"hilo::NeuronInstCombine": 0.0008500000112690032,
"hilo::NeuronOpFusion": 0.00020799999765586108,
"hilo::ReplaceTokenTypeWithU8Pass": 0.0007050000131130219,
"hilo::ScheduleFusion": 4.3000000005122274e-05,
"hilo::SixtyFourHack": 0.0005150000215508044,
"hilo::VerifyAliasing": 0.00035099999513477087,
"hlo-mac-count": 0.08947300165891647,
"io-con-pipe-begin": 6.199999916134402e-05,
"io-con-pipe-end": 9.999999974752427e-07,
"io-layout-normalization": 0.0706389993429184,
"legalize-ccops-for-tensorizer": 0.00013800000306218863,
"legalize-compare": 0.00027200000477023423,
"lower-argminmax-custom-call": 0.0004130000015720725,
"map-inline": 0.0010939999483525753,
"metadata-naming": 0.009581999853253365,
"mlir::detail::OpToOpPassAdaptor": 0.0004569999873638153,
"mlir::hlo::MhloToPyPenguin": 0.5751969814300537,
"mlir::mhlo::LowerComplexExtraPass": 0.00508299982175231,
"mlir::mhlo::LowerComplexPass": 0.0012469999492168427,
"native-to-custom-softmax": 0.0004349999944679439,
"native-to-custom-softmax-dx": 0.00044800000614486635,
"neuron-hlo-verifier": 0.3175100088119507,
"operand_upcaster": 0.0018560000462457538,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 0.0,
"post-partition-simplification": 0.48272499442100525,
"pre-hlo-begin": 1.4000000192027073e-05,
"pre-hlo-end": 1.9999999949504854e-06,
"replace-minimum-constant": 0.00022400000307243317,
"reshape-mover": 5.700000110664405e-05,
"simplify-concat": 0.002764000091701746,
"simplify-while-loops": 9.40000027185306e-05,
"transform-variadic-reduce": 0.001218999968841672,
"tuple-simplifier": 0.00048099999548867345,
"unpack-nested-aws-ntwsr": 0.0007040000054985285,
"unroll-while-loop": 1.2000000424450263e-05
},
"hilo": {
"HloMacCount": 1796005888.0,
"Traffic": 1252622464.0
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 45730,
"StaticProfiler::AifUb": 19.087068557739258,
"StaticProfiler::ArithmeticIntensityTensorizer": 29.72186851501465,
"StaticProfiler::AverageDmaLength": 2694.496826171875,
"StaticProfiler::DDRTransferBytes": 954289776,
"StaticProfiler::InternalTransferBytes": 196796000,
"StaticProfiler::LoadExpanded": 281500,
"StaticProfiler::StoreExpanded": 4273,
"StaticProfiler::TotalDMAExpanded": 285773,
"StaticProfiler::TotalDynamicInstancesCount": 62161,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 53675,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 123,
"TilingProfiler::MatMultInstructionsAfterTiling": 32320,
"TilingProfiler::NumPfTransposes": 292,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 142,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 8762,
"TilingProfiler::PfTransposeInstructionsForIo": 6564,
"TilingProfiler::PfTransposeInstructionsForLocal": 536,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
"TilingProfiler::ReduceInstructionsAfterTiling": 171,
"TilingProfiler::SimdInstructionsAfterTiling": 2465,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"all": {
"compiletime": {
"CanonicalizeConv": 0.0,
"CanonicalizeForTensorizer": 0.0005849999724887311,
"Canonicalizer": 0.015991000458598137,
"HoistCompute": 9.899999713525176e-05,
"IdentifyCrossPassTensors": 0.00011500000255182385,
"MemcastMotion": 0.0002789999998640269,
"PenguinizeFunctions": 0.0005789999850094318,
"PruneFunctions": 0.00018899999849963933,
"RemoveOptimizationBarriers": 0.0004290000069886446,
"ScatterMotion": 0.005127000156790018,
"TensorizerLegalizationPass": 0.000455000001238659,
"VerifySupportedOps": 0.0002469999890308827,
"algsimp": 0.0018210000125691295,
"batchnorm_expander": 0.0022700000554323196,
"boundary-marker-removal": 0.0009800000116229057,
"call-inliner": 0.0002410000015515834,
"canonicalize-boundary-marker": 0.0009430000209249556,
"collective-stream-id-checker": 0.0008960000122897327,
"comparison-expander": 0.000623999978415668,
"computation-deduplicator": 0.003625999903306365,
"config-lowering": 0.00017899999511428177,
"constant_folding": 0.0668639987707138,
"cse": 0.001184999942779541,
"dce": 6.399999983841553e-05,
"dynamic-slice-transpose": 0.0003530000103637576,
"eliminate-redundant-compare": 0.00014400000509340316,
"emit-offloaded-dropout": 0.0007820000173524022,
"flatten-call-graph": 0.0009519999730400741,
"fuse-send-recv": 0.07950299978256226,
"hilo-conditional-to-select": 0.00015700000221841037,
"hilo::LegalizeAlias": 0.003444999922066927,
"hilo::NeuronInstCombine": 0.0008500000112690032,
"hilo::NeuronOpFusion": 0.00020799999765586108,
"hilo::ReplaceTokenTypeWithU8Pass": 0.0007050000131130219,
"hilo::ScheduleFusion": 4.3000000005122274e-05,
"hilo::SixtyFourHack": 0.0005150000215508044,
"hilo::VerifyAliasing": 0.00035099999513477087,
"hlo-mac-count": 0.08947300165891647,
"io-con-pipe-begin": 6.199999916134402e-05,
"io-con-pipe-end": 9.999999974752427e-07,
"io-layout-normalization": 0.0706389993429184,
"legalize-ccops-for-tensorizer": 0.00013800000306218863,
"legalize-compare": 0.00027200000477023423,
"lower-argminmax-custom-call": 0.0004130000015720725,
"map-inline": 0.0010939999483525753,
"metadata-naming": 0.009581999853253365,
"mlir::detail::OpToOpPassAdaptor": 0.0004569999873638153,
"mlir::hlo::MhloToPyPenguin": 0.5751969814300537,
"mlir::mhlo::LowerComplexExtraPass": 0.00508299982175231,
"mlir::mhlo::LowerComplexPass": 0.0012469999492168427,
"native-to-custom-softmax": 0.0004349999944679439,
"native-to-custom-softmax-dx": 0.00044800000614486635,
"neuron-hlo-verifier": 0.3175100088119507,
"operand_upcaster": 0.0018560000462457538,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 0.0,
"post-partition-simplification": 0.48272499442100525,
"pre-hlo-begin": 1.4000000192027073e-05,
"pre-hlo-end": 1.9999999949504854e-06,
"replace-minimum-constant": 0.00022400000307243317,
"reshape-mover": 5.700000110664405e-05,
"simplify-concat": 0.002764000091701746,
"simplify-while-loops": 9.40000027185306e-05,
"transform-variadic-reduce": 0.001218999968841672,
"tuple-simplifier": 0.00048099999548867345,
"unpack-nested-aws-ntwsr": 0.0007040000054985285,
"unroll-while-loop": 1.2000000424450263e-05
}
},
"cumsum": {
"compiletime": {
"CoalesceCCOp": 0.0002155303955078125,
"DMALocalityOpt": 0.00016641616821289063,
"DMAProfiler": 0.0007414817810058594,
"DataStreaming": 0.0002970695495605469,
"DoNothing": 0.0001289844512939453,
"ExpandISAMacro": 0.0005061626434326172,
"FactorizeBlkDims": 0.0004284381866455078,
"InferPSumTensor": 0.0005972385406494141,
"InferSharedMemLoc": 0.0002605915069580078,
"InsertCoreBarrier": 0.000263214111328125,
"LateLegalizeInst": 0.00037741661071777344,
"LateNeuronInstComb": 0.0005505084991455078,
"LegalizeSundaAccess": 0.0013835430145263672,
"LegalizeType": 0.00022840499877929688,
"LowerBroadcast": 0.0002124309539794922,
"LowerIntrinsics": 0.0002105236053466797,
"LowerTranspose": 0.00021958351135253906,
"NeuronInstComb": 0.000583648681640625,
"NeuronLICM": 0.00035643577575683594,
"NeuronSimplifyPredicates": 0.002285003662109375,
"NeuronValueNumbering": 0.0004379749298095703,
"SFKVectorizer": 0.0024003982543945313,
"SimpleAllReduceTiling": 0.00020265579223632813,
"SimplifyNeuronTensor": 0.0004928112030029297,
"SpillPSum": 0.00045371055603027344,
"WeightCoalescing": 0.0002079010009765625
}
},
"sg00": {
"hilo": {
"ArithmeticIntensity": 2.867593288421631,
"HloMacCount": 1796005888.0,
"Traffic": 1252622464.0
}
},
"sg0000": {
"compiletime": {
"AGOrderingAnalysisPass": 6.775559425354004,
"AffinePredicateResolution": 0.7124242782592773,
"AliasDependencyElimination": 0.0037508010864257813,
"AliasDependencyInduction": 17.88080596923828,
"AliasDependencyReset": 19.01345443725586,
"BFComputeCutting": 0.14321279525756836,
"BirCodeGenLoop": 4.049878120422363,
"CCOpFusion": 0.9675219058990479,
"CanonicalizeDAGForPGTiling": 0.18202924728393555,
"CanonicalizeIR": 0.8085732460021973,
"CoalesceCCOp": 0.2558913230895996,
"CommuteConcat": 0.027615070343017578,
"DMALocalityOpt": 0.04517340660095215,
"DMAProfiler": 0.08888816833496094,
"DMATilingProfiler": 0.09855914115905762,
"DataLocalityOpt": 2.8309335708618164,
"DataStreaming": 0.29266810417175293,
"DeConcat": 0.08213448524475098,
"DeadCodeElimination": 0.030428647994995117,
"DeadStoreElimination": 1.051145315170288,
"DelinearIndices": 0.5529322624206543,
"Delinearization": 0.14493751525878906,
"DelinearizeSPMD": 0.181105375289917,
"DoNothing": 7.724761962890625e-05,
"DramToDramTranspose": 0.5176758766174316,
"DumpGraphAndMetadata": 0.1749565601348877,
"EliminateDivs": 1.6087908744812012,
"ExpandBatchNorm": 1.096595048904419,
"ExpandISAMacro": 0.08580923080444336,
"FactorizeBlkDims": 0.49906015396118164,
"FactorizeThreadAxesInFreeDims": 0.08890652656555176,
"FlattenMacroLoop": 0.08152222633361816,
"GenericAccessSimplifier": 0.02545022964477539,
"InferInitValue": 1.2978432178497314,
"InferIntrinsicOnCC": 0.7616860866546631,
"InferNeuronTensor": 2.6053829193115234,
"InferNonlocalTensors": 7.113053321838379,
"InferPSumTensor": 4.437821388244629,
"InferShardAxis": 12.849608421325684,
"InferSharedMemLoc": 0.18901801109313965,
"InlineNativeKernels": 0.08442234992980957,
"InsertCoreBarrier": 0.1489267349243164,
"InsertIOTransposes": 0.8695223331451416,
"InsertImplicitShardAxisBeforeISel": 0.49620485305786133,
"InsertLocalTransposes": 1.934779167175293,
"InsertOffloadedTransposes": 0.23423099517822266,
"LICM": 0.1213834285736084,
"LateLegalizeInst": 0.2814202308654785,
"LateLegalizePostSplit": 0.22411513328552246,
"LateLowerReshapeOp": 0.033789873123168945,
"LateLowerTensorOp": 4.297260761260986,
"LateNeuronInstComb": 1.014098882675171,
"LayoutPreprocessing": 1.8572075366973877,
"LayoutPreprocessingAndAnalysis": 2.2904410362243652,
"LayoutRequirementAnalysis": 0.4237644672393799,
"LegalizeCCOpLayout": 1.1935856342315674,
"LegalizeOpLevelAlias": 0.7793729305267334,
"LegalizePartitionReduce": 0.09432244300842285,
"LegalizeSundaAccess": 0.972252607345581,
"LegalizeSundaMacro": 0.7198967933654785,
"LegalizeType": 0.16599535942077637,
"LocalLayoutOpt": 0.6666848659515381,
"LoopFusion": 0.29988932609558105,
"LoopSplitting": 0.05208778381347656,
"LowerBroadcast": 0.07862257957458496,
"LowerCCOpBlockAxis": 0.2862672805786133,
"LowerComplexBroadcast": 0.09356045722961426,
"LowerIntrinsics": 1.3567829132080078,
"LowerShardAxis": 0.39397311210632324,
"LowerTensorOp": 4.018380165100098,
"LowerToSendRecv": 0.27350616455078125,
"LowerTranspose": 0.513293981552124,
"MacroGeneration": 2.0922586917877197,
"MaskPropagation": 0.12857651710510254,
"MemcpyElimination": 30.280338287353516,
"MutateDataType": 0.03657245635986328,
"NeuronAliasDependencyInduction": 0.026292085647583008,
"NeuronAliasDependencyReset": 0.03253936767578125,
"NeuronInstComb": 0.35875773429870605,
"NeuronLICM": 0.29233455657958984,
"NeuronLoopFusion": 1.711080551147461,
"NeuronLoopInterchange": 0.08134770393371582,
"NeuronSimplifier": 0.5104148387908936,
"NeuronSimplifyPredicates": 0.2326216697692871,
"NeuronValueNumbering": 0.11057472229003906,
"OptimizeAliasedCopyChain": 0.40732264518737793,
"OptimizeNKIKernels": 1.0493371486663818,
"PAGLayoutOpt": 18.23680305480957,
"PComputeCutting": 0.5901892185211182,
"PGLayoutTilingPipeline": 54.30891036987305,
"PGTiling": 10.472368240356445,
"PadElimination": 0.0187380313873291,
"ParAxesAnnotation": 16.293506622314453,
"PartialLoopFusion": 1.5408625602722168,
"PartialSimdFusion": 0.8349573612213135,
"PerfectLoopNest": 0.07371997833251953,
"RecognizeOpIdiom": 0.12681221961975098,
"Recompute": 0.007548809051513672,
"RelaxPredicates": 0.11768078804016113,
"Rematerialization": 0.15080475807189941,
"RemoveShardedPartitionAxes": 1.1734652519226074,
"ReshapeWeights": 0.022508859634399414,
"ResolveAccessConflict": 0.20145153999328613,
"ResolveComplicatePredicates": 0.3191838264465332,
"RewriteReplicationMatmul": 0.039293527603149414,
"RewriteWeights": 0.0773766040802002,
"SFKVectorizer": 11.047651290893555,
"ShardingPropagationAnalysis": 0.665086030960083,
"SimpleAllReduceTiling": 0.07064080238342285,
"Simplifier": 0.09160900115966797,
"SimplifyMacroPredicates": 0.28479433059692383,
"SimplifyNeuronTensor": 0.4839820861816406,
"SimplifySlice": 0.02684783935546875,
"SimplifyTensor": 0.2644212245941162,
"SpillPSum": 1.035245418548584,
"SplitAPUnionSets": 1.0491511821746826,
"SplitAccGrp": 0.055588722229003906,
"StaticProfiler": 0.39172840118408203,
"StaticTransposeLocalTensor": 0.7211995124816895,
"SundaISel": 1.5357580184936523,
"TCTransform": 0.029030561447143555,
"TensorInitialization": 0.20336008071899414,
"TensorOpSimplifier": 3.100177049636841,
"TensorOpTransform": 20.175758361816406,
"TileCCOps": 0.17463970184326172,
"TilingProfiler": 0.7014575004577637,
"TransformConvOp": 0.9923229217529297,
"TritiumFusion": 0.33277034759521484,
"ValueNumbering": 0.08796954154968262,
"VectorizeDMA": 0.520815372467041,
"VectorizeMatMult": 0.08555078506469727,
"WeightCoalescing": 0.08340644836425781,
"ZeroSizeTensorElimination": 0.0006709098815917969
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 45730,
"StaticProfiler::AifUb": 19.087068557739258,
"StaticProfiler::ArithmeticIntensityTensorizer": 29.72186851501465,
"StaticProfiler::AverageDmaLength": 2694.496826171875,
"StaticProfiler::AverageFractalPeUtilization": 97.78141784667969,
"StaticProfiler::AveragePartitionUtilization": 90.33787536621094,
"StaticProfiler::AveragePeUtilization": 81.002685546875,
"StaticProfiler::DDRTransferBytes": 954289776,
"StaticProfiler::InternalTransferBytes": 196796000,
"StaticProfiler::LoadExpanded": 281500,
"StaticProfiler::LocalizationEfficiency": 155.71730041503906,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 160.65768432617188,
"StaticProfiler::StoreExpanded": 4273,
"StaticProfiler::TotalDMAExpanded": 285773,
"StaticProfiler::TotalDynamicInstancesCount": 62161,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 53675,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 123,
"TilingProfiler::MatMultInstructionsAfterTiling": 32320,
"TilingProfiler::NumPfTransposes": 292,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 142,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 8762,
"TilingProfiler::PfTransposeInstructionsForIo": 6564,
"TilingProfiler::PfTransposeInstructionsForLocal": 536,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
"TilingProfiler::ReduceInstructionsAfterTiling": 171,
"TilingProfiler::SimdInstructionsAfterTiling": 2465,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"topk": {
"compiletime": {
"CoalesceCCOp": 0.003664731979370117,
"DMALocalityOpt": 0.002711057662963867,
"DMAProfiler": 0.003767251968383789,
"DataStreaming": 0.0067331790924072266,
"DoNothing": 0.00014734268188476563,
"ExpandISAMacro": 0.006513833999633789,
"FactorizeBlkDims": 0.012669801712036133,
"InferPSumTensor": 0.012490510940551758,
"InferSharedMemLoc": 0.0029754638671875,
"InsertCoreBarrier": 0.00330352783203125,
"LateLegalizeInst": 0.007447957992553711,
"LateNeuronInstComb": 0.008418560028076172,
"LegalizeSundaAccess": 0.02680182456970215,
"LegalizeType": 0.010413169860839844,
"LowerBroadcast": 0.0033206939697265625,
"LowerIntrinsics": 0.003704547882080078,
"LowerTranspose": 0.0035905838012695313,
"NeuronInstComb": 0.008339166641235352,
"NeuronLICM": 0.009230852127075195,
"NeuronSimplifyPredicates": 0.006308555603027344,
"NeuronValueNumbering": 0.003956317901611328,
"SFKVectorizer": 0.03573727607727051,
"SimpleAllReduceTiling": 0.0036857128143310547,
"SimplifyNeuronTensor": 0.054434776306152344,
"SpillPSum": 0.02359294891357422,
"WeightCoalescing": 0.0065898895263671875
}
}
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2e9104a1923d9a888389573352849a655d6f7005c5782380d0fa3306dab803f3
size 3390464

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a4a34ac27ace8c75df52c6f7fb5e0c1736f327638e3ac42ff906c6b52e125a5e
size 2464555

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0baed4147c67e64b22e5d725b3e55d70e3e16ed2fe838758927f1cb592ebdc7a
size 2550451

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2e9104a1923d9a888389573352849a655d6f7005c5782380d0fa3306dab803f3
size 3390464

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 4,
"bucket_n_active_tokens": false,
"buckets": [
512
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 1,
"chunked_prefill_config": null,
"context_encoding_buckets": null,
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": false,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 1,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": [
512
],
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_02e08d5fdadfbcb8a569+a6f96e4f.hlo_module.pb --output model.MODULE_02e08d5fdadfbcb8a569+a6f96e4f.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk3/log-neuron-cc.txt"]

View File

@@ -0,0 +1,590 @@
{
"Average": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 97.88961791992188,
"StaticProfiler::AveragePartitionUtilization": 89.79463958740234,
"StaticProfiler::AveragePeUtilization": 78.904052734375,
"StaticProfiler::LocalizationEfficiency": 146.6905975341797,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 151.06674194335938,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0
}
},
"Count": {
"tensorizer": {
"StaticProfiler::AverageFractalPeUtilization": 1,
"StaticProfiler::AveragePartitionUtilization": 1,
"StaticProfiler::AveragePeUtilization": 1,
"StaticProfiler::LocalizationEfficiency": 1,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 1,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 1,
"TilingProfiler::AveragePeUtilizationAfterTiling": 1
}
},
"Sum": {
"compiletime": {
"AGOrderingAnalysisPass": 6.388453960418701,
"AffinePredicateResolution": 0.40378904342651367,
"AliasDependencyElimination": 0.016310930252075195,
"AliasDependencyInduction": 15.282819747924805,
"AliasDependencyReset": 16.618453979492188,
"BFComputeCutting": 0.1391315460205078,
"BirCodeGenLoop": 4.402980804443359,
"CCOpFusion": 0.8881416320800781,
"CanonicalizeConv": 4.999999873689376e-06,
"CanonicalizeDAGForPGTiling": 0.20721173286437988,
"CanonicalizeForTensorizer": 0.00076299998909235,
"CanonicalizeIR": 1.028043270111084,
"Canonicalizer": 0.08714199811220169,
"CoalesceCCOp": 0.2357621192932129,
"CommuteConcat": 0.0287783145904541,
"DMALocalityOpt": 0.07430553436279297,
"DMAProfiler": 0.08057951927185059,
"DMATilingProfiler": 0.09187006950378418,
"DataLocalityOpt": 2.820284605026245,
"DataStreaming": 0.4295690059661865,
"DeConcat": 0.06640625,
"DeadCodeElimination": 0.053946495056152344,
"DeadStoreElimination": 1.0540196895599365,
"DelinearIndices": 0.3954615592956543,
"Delinearization": 0.13945698738098145,
"DelinearizeSPMD": 0.17487859725952148,
"DoNothing": 0.0004494190216064453,
"DramToDramTranspose": 0.34574389457702637,
"DumpGraphAndMetadata": 0.23639750480651855,
"EliminateDivs": 1.4790117740631104,
"ExpandBatchNorm": 1.123610019683838,
"ExpandISAMacro": 0.12250304222106934,
"FactorizeBlkDims": 0.49267148971557617,
"FactorizeThreadAxesInFreeDims": 0.08273863792419434,
"FlattenMacroLoop": 0.08472108840942383,
"GenericAccessSimplifier": 0.025115966796875,
"HoistCompute": 0.0,
"IdentifyCrossPassTensors": 0.0003549999964889139,
"InferInitValue": 1.3054986000061035,
"InferIntrinsicOnCC": 0.7679235935211182,
"InferNeuronTensor": 1.671952724456787,
"InferNonlocalTensors": 7.110470771789551,
"InferPSumTensor": 4.2829790115356445,
"InferShardAxis": 12.786405563354492,
"InferSharedMemLoc": 0.16948676109313965,
"InlineNativeKernels": 0.05566692352294922,
"InsertCoreBarrier": 0.15158963203430176,
"InsertIOTransposes": 2.3931386470794678,
"InsertImplicitShardAxisBeforeISel": 0.5525217056274414,
"InsertLocalTransposes": 0.7429313659667969,
"InsertOffloadedTransposes": 0.19022369384765625,
"LICM": 0.14770793914794922,
"LateLegalizeInst": 0.31015753746032715,
"LateLegalizePostSplit": 0.22165155410766602,
"LateLowerReshapeOp": 0.03522944450378418,
"LateLowerTensorOp": 6.813569068908691,
"LateNeuronInstComb": 1.2720398902893066,
"LayoutPreprocessing": 1.8736376762390137,
"LayoutPreprocessingAndAnalysis": 2.3256783485412598,
"LayoutRequirementAnalysis": 0.4448714256286621,
"LegalizeCCOpLayout": 1.317772388458252,
"LegalizeOpLevelAlias": 0.5106728076934814,
"LegalizePartitionReduce": 0.08073091506958008,
"LegalizeSundaAccess": 0.9463906288146973,
"LegalizeSundaMacro": 0.7451527118682861,
"LegalizeType": 0.13950800895690918,
"LocalLayoutOpt": 0.6608095169067383,
"LoopFusion": 0.3101928234100342,
"LoopSplitting": 0.036502838134765625,
"LowerBroadcast": 0.06515121459960938,
"LowerCCOpBlockAxis": 0.32977890968322754,
"LowerComplexBroadcast": 0.15041875839233398,
"LowerIntrinsics": 1.0637662410736084,
"LowerShardAxis": 0.4116206169128418,
"LowerTensorOp": 3.285423755645752,
"LowerToSendRecv": 0.25904369354248047,
"LowerTranspose": 0.5238301753997803,
"MacroGeneration": 2.1192569732666016,
"MaskPropagation": 0.14647436141967773,
"MemcastMotion": 0.0,
"MemcpyElimination": 29.3009090423584,
"MutateDataType": 0.037621498107910156,
"NeuronAliasDependencyInduction": 0.06313967704772949,
"NeuronAliasDependencyReset": 0.08403873443603516,
"NeuronInstComb": 0.3617970943450928,
"NeuronLICM": 0.2975320816040039,
"NeuronLoopFusion": 1.491703987121582,
"NeuronLoopInterchange": 0.06900668144226074,
"NeuronSimplifier": 0.5315425395965576,
"NeuronSimplifyPredicates": 0.2570762634277344,
"NeuronValueNumbering": 0.11507940292358398,
"OptimizeAliasedCopyChain": 0.6118557453155518,
"OptimizeNKIKernels": 1.0786433219909668,
"PAGLayoutOpt": 18.570398330688477,
"PComputeCutting": 0.6222004890441895,
"PGLayoutTilingPipeline": 55.62285614013672,
"PGTiling": 10.0595703125,
"PadElimination": 0.021188735961914063,
"ParAxesAnnotation": 17.818796157836914,
"PartialLoopFusion": 1.5088038444519043,
"PartialSimdFusion": 0.8573262691497803,
"PenguinizeFunctions": 0.000754999986384064,
"PerfectLoopNest": 0.07143402099609375,
"PruneFunctions": 0.0012410000199452043,
"RecognizeOpIdiom": 0.1291654109954834,
"Recompute": 0.007833480834960938,
"RelaxPredicates": 0.14508056640625,
"Rematerialization": 0.17157793045043945,
"RemoveOptimizationBarriers": 0.0005370000144466758,
"RemoveShardedPartitionAxes": 1.3566563129425049,
"ReshapeWeights": 0.022662878036499023,
"ResolveAccessConflict": 0.2033238410949707,
"ResolveComplicatePredicates": 0.49439358711242676,
"RewriteReplicationMatmul": 0.03938794136047363,
"RewriteWeights": 0.07809877395629883,
"SFKVectorizer": 10.350098609924316,
"ScatterMotion": 0.0,
"ShardingPropagationAnalysis": 0.7199399471282959,
"SimpleAllReduceTiling": 0.07370853424072266,
"Simplifier": 0.09367704391479492,
"SimplifyMacroPredicates": 0.28262877464294434,
"SimplifyNeuronTensor": 0.9568257331848145,
"SimplifySlice": 0.027447223663330078,
"SimplifyTensor": 0.2734675407409668,
"SpillPSum": 1.2486350536346436,
"SplitAPUnionSets": 0.8627588748931885,
"SplitAccGrp": 0.1022341251373291,
"StaticProfiler": 0.4753732681274414,
"StaticTransposeLocalTensor": 0.6612956523895264,
"SundaISel": 1.6735970973968506,
"TCTransform": 0.029105186462402344,
"TensorInitialization": 0.1944262981414795,
"TensorOpSimplifier": 3.3376035690307617,
"TensorOpTransform": 19.926769256591797,
"TensorizerLegalizationPass": 0.0006489999941550195,
"TileCCOps": 0.1856687068939209,
"TilingProfiler": 0.5146362781524658,
"TransformConvOp": 1.4806337356567383,
"TritiumFusion": 0.2828052043914795,
"ValueNumbering": 0.08587479591369629,
"VectorizeDMA": 0.5646088123321533,
"VectorizeMatMult": 0.08723974227905273,
"VerifySupportedOps": 0.0004870000120718032,
"WeightCoalescing": 0.06492829322814941,
"ZeroSizeTensorElimination": 0.0008580684661865234,
"algsimp": 0.0014750000555068254,
"batchnorm_expander": 0.0015930000226944685,
"boundary-marker-removal": 0.0007229999755509198,
"call-inliner": 0.00018899999849963933,
"canonicalize-boundary-marker": 0.0011119999689981341,
"collective-stream-id-checker": 8.900000102585182e-05,
"comparison-expander": 0.0007440000190399587,
"computation-deduplicator": 0.0037249999586492777,
"config-lowering": 0.06129100173711777,
"constant_folding": 0.00012700000661425292,
"cse": 0.0007800000021234155,
"dce": 5.199999941396527e-05,
"dynamic-slice-transpose": 0.0002530000056140125,
"eliminate-redundant-compare": 0.0001140000022132881,
"emit-offloaded-dropout": 0.0006549999816343188,
"flatten-call-graph": 0.0013830000534653664,
"fuse-send-recv": 0.07738199830055237,
"hilo-conditional-to-select": 0.00014400000509340316,
"hilo::LegalizeAlias": 0.004616000223904848,
"hilo::NeuronInstCombine": 0.0015979999443516135,
"hilo::NeuronOpFusion": 0.0001250000059371814,
"hilo::ReplaceTokenTypeWithU8Pass": 0.0016670000040903687,
"hilo::ScheduleFusion": 9.999999747378752e-06,
"hilo::SixtyFourHack": 0.0008529999759048223,
"hilo::VerifyAliasing": 0.000291000003926456,
"hlo-mac-count": 0.02539600059390068,
"io-con-pipe-begin": 7.999999979801942e-06,
"io-con-pipe-end": 0.0,
"io-layout-normalization": 0.0035389999393373728,
"legalize-ccops-for-tensorizer": 0.00014200000441633165,
"legalize-compare": 0.0007239999831654131,
"lower-argminmax-custom-call": 0.0003229999856557697,
"map-inline": 0.0010969999711960554,
"metadata-naming": 0.009553000330924988,
"mlir::detail::OpToOpPassAdaptor": 9.999999747378752e-05,
"mlir::hlo::MhloToPyPenguin": 0.6092489957809448,
"mlir::mhlo::LowerComplexExtraPass": 0.0049149999395012856,
"mlir::mhlo::LowerComplexPass": 0.008914999663829803,
"native-to-custom-softmax": 0.0010550000006332994,
"native-to-custom-softmax-dx": 0.001585999969393015,
"neuron-hlo-verifier": 0.22506099939346313,
"operand_upcaster": 0.003352999920025468,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 0.0,
"post-partition-simplification": 0.37791401147842407,
"pre-hlo-begin": 3.999999989900971e-06,
"pre-hlo-end": 9.999999974752427e-07,
"replace-minimum-constant": 0.00019299999985378236,
"reshape-mover": 5.2999999752501026e-05,
"simplify-concat": 0.06822799891233444,
"simplify-while-loops": 5.0999999075429514e-05,
"transform-variadic-reduce": 0.0008849999867379665,
"tuple-simplifier": 0.00012099999730708078,
"unpack-nested-aws-ntwsr": 0.000615999975707382,
"unroll-while-loop": 7.000000096013537e-06
},
"hilo": {
"HloMacCount": 1854726144.0,
"Traffic": 1252630656.0
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 51674,
"StaticProfiler::AifUb": 21.735599517822266,
"StaticProfiler::ArithmeticIntensityTensorizer": 31.884082794189453,
"StaticProfiler::AverageDmaLength": 2843.9130859375,
"StaticProfiler::DDRTransferBytes": 1013018224,
"StaticProfiler::InternalTransferBytes": 226623072,
"StaticProfiler::LoadExpanded": 281500,
"StaticProfiler::StoreExpanded": 4273,
"StaticProfiler::TotalDMAExpanded": 285773,
"StaticProfiler::TotalDynamicInstancesCount": 67989,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 57540,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 123,
"TilingProfiler::MatMultInstructionsAfterTiling": 35008,
"TilingProfiler::NumPfTransposes": 292,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 142,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 10670,
"TilingProfiler::PfTransposeInstructionsForIo": 8360,
"TilingProfiler::PfTransposeInstructionsForLocal": 648,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
"TilingProfiler::ReduceInstructionsAfterTiling": 283,
"TilingProfiler::SimdInstructionsAfterTiling": 2805,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"all": {
"compiletime": {
"CanonicalizeConv": 4.999999873689376e-06,
"CanonicalizeForTensorizer": 0.00076299998909235,
"Canonicalizer": 0.08714199811220169,
"HoistCompute": 0.0,
"IdentifyCrossPassTensors": 0.0003549999964889139,
"MemcastMotion": 0.0,
"PenguinizeFunctions": 0.000754999986384064,
"PruneFunctions": 0.0012410000199452043,
"RemoveOptimizationBarriers": 0.0005370000144466758,
"ScatterMotion": 0.0,
"TensorizerLegalizationPass": 0.0006489999941550195,
"VerifySupportedOps": 0.0004870000120718032,
"algsimp": 0.0014750000555068254,
"batchnorm_expander": 0.0015930000226944685,
"boundary-marker-removal": 0.0007229999755509198,
"call-inliner": 0.00018899999849963933,
"canonicalize-boundary-marker": 0.0011119999689981341,
"collective-stream-id-checker": 8.900000102585182e-05,
"comparison-expander": 0.0007440000190399587,
"computation-deduplicator": 0.0037249999586492777,
"config-lowering": 0.06129100173711777,
"constant_folding": 0.00012700000661425292,
"cse": 0.0007800000021234155,
"dce": 5.199999941396527e-05,
"dynamic-slice-transpose": 0.0002530000056140125,
"eliminate-redundant-compare": 0.0001140000022132881,
"emit-offloaded-dropout": 0.0006549999816343188,
"flatten-call-graph": 0.0013830000534653664,
"fuse-send-recv": 0.07738199830055237,
"hilo-conditional-to-select": 0.00014400000509340316,
"hilo::LegalizeAlias": 0.004616000223904848,
"hilo::NeuronInstCombine": 0.0015979999443516135,
"hilo::NeuronOpFusion": 0.0001250000059371814,
"hilo::ReplaceTokenTypeWithU8Pass": 0.0016670000040903687,
"hilo::ScheduleFusion": 9.999999747378752e-06,
"hilo::SixtyFourHack": 0.0008529999759048223,
"hilo::VerifyAliasing": 0.000291000003926456,
"hlo-mac-count": 0.02539600059390068,
"io-con-pipe-begin": 7.999999979801942e-06,
"io-con-pipe-end": 0.0,
"io-layout-normalization": 0.0035389999393373728,
"legalize-ccops-for-tensorizer": 0.00014200000441633165,
"legalize-compare": 0.0007239999831654131,
"lower-argminmax-custom-call": 0.0003229999856557697,
"map-inline": 0.0010969999711960554,
"metadata-naming": 0.009553000330924988,
"mlir::detail::OpToOpPassAdaptor": 9.999999747378752e-05,
"mlir::hlo::MhloToPyPenguin": 0.6092489957809448,
"mlir::mhlo::LowerComplexExtraPass": 0.0049149999395012856,
"mlir::mhlo::LowerComplexPass": 0.008914999663829803,
"native-to-custom-softmax": 0.0010550000006332994,
"native-to-custom-softmax-dx": 0.001585999969393015,
"neuron-hlo-verifier": 0.22506099939346313,
"operand_upcaster": 0.003352999920025468,
"post-par-pipe-begin": 9.999999974752427e-07,
"post-par-pipe-end": 0.0,
"post-partition-simplification": 0.37791401147842407,
"pre-hlo-begin": 3.999999989900971e-06,
"pre-hlo-end": 9.999999974752427e-07,
"replace-minimum-constant": 0.00019299999985378236,
"reshape-mover": 5.2999999752501026e-05,
"simplify-concat": 0.06822799891233444,
"simplify-while-loops": 5.0999999075429514e-05,
"transform-variadic-reduce": 0.0008849999867379665,
"tuple-simplifier": 0.00012099999730708078,
"unpack-nested-aws-ntwsr": 0.000615999975707382,
"unroll-while-loop": 7.000000096013537e-06
}
},
"cumsum": {
"compiletime": {
"CoalesceCCOp": 0.00023365020751953125,
"DMALocalityOpt": 0.00019097328186035156,
"DMAProfiler": 0.0009503364562988281,
"DataStreaming": 0.00032329559326171875,
"DoNothing": 0.00015044212341308594,
"ExpandISAMacro": 0.0005333423614501953,
"FactorizeBlkDims": 0.0005185604095458984,
"InferPSumTensor": 0.0006201267242431641,
"InferSharedMemLoc": 0.0002682209014892578,
"InsertCoreBarrier": 0.0003275871276855469,
"LateLegalizeInst": 0.0004162788391113281,
"LateNeuronInstComb": 0.0005996227264404297,
"LegalizeSundaAccess": 0.0016045570373535156,
"LegalizeType": 0.0002541542053222656,
"LowerBroadcast": 0.00023365020751953125,
"LowerIntrinsics": 0.00023651123046875,
"LowerTranspose": 0.0002307891845703125,
"NeuronInstComb": 0.0006008148193359375,
"NeuronLICM": 0.00036597251892089844,
"NeuronSimplifyPredicates": 0.002484560012817383,
"NeuronValueNumbering": 0.0004477500915527344,
"SFKVectorizer": 0.002768278121948242,
"SimpleAllReduceTiling": 0.00021409988403320313,
"SimplifyNeuronTensor": 0.0005936622619628906,
"SpillPSum": 0.0005390644073486328,
"WeightCoalescing": 0.00023293495178222656
}
},
"sg00": {
"hilo": {
"ArithmeticIntensity": 2.961329698562622,
"HloMacCount": 1854726144.0,
"Traffic": 1252630656.0
}
},
"sg0000": {
"compiletime": {
"AGOrderingAnalysisPass": 6.388453960418701,
"AffinePredicateResolution": 0.40378904342651367,
"AliasDependencyElimination": 0.016310930252075195,
"AliasDependencyInduction": 15.282819747924805,
"AliasDependencyReset": 16.618453979492188,
"BFComputeCutting": 0.1391315460205078,
"BirCodeGenLoop": 4.402980804443359,
"CCOpFusion": 0.8881416320800781,
"CanonicalizeDAGForPGTiling": 0.20721173286437988,
"CanonicalizeIR": 1.028043270111084,
"CoalesceCCOp": 0.23186254501342773,
"CommuteConcat": 0.0287783145904541,
"DMALocalityOpt": 0.07128620147705078,
"DMAProfiler": 0.07605504989624023,
"DMATilingProfiler": 0.09187006950378418,
"DataLocalityOpt": 2.820284605026245,
"DataStreaming": 0.42258405685424805,
"DeConcat": 0.06640625,
"DeadCodeElimination": 0.053946495056152344,
"DeadStoreElimination": 1.0540196895599365,
"DelinearIndices": 0.3954615592956543,
"Delinearization": 0.13945698738098145,
"DelinearizeSPMD": 0.17487859725952148,
"DoNothing": 0.00014019012451171875,
"DramToDramTranspose": 0.34574389457702637,
"DumpGraphAndMetadata": 0.23639750480651855,
"EliminateDivs": 1.4790117740631104,
"ExpandBatchNorm": 1.123610019683838,
"ExpandISAMacro": 0.11814379692077637,
"FactorizeBlkDims": 0.4797940254211426,
"FactorizeThreadAxesInFreeDims": 0.08273863792419434,
"FlattenMacroLoop": 0.08472108840942383,
"GenericAccessSimplifier": 0.025115966796875,
"InferInitValue": 1.3054986000061035,
"InferIntrinsicOnCC": 0.7679235935211182,
"InferNeuronTensor": 1.671952724456787,
"InferNonlocalTensors": 7.110470771789551,
"InferPSumTensor": 4.2699432373046875,
"InferShardAxis": 12.786405563354492,
"InferSharedMemLoc": 0.16634631156921387,
"InlineNativeKernels": 0.05566692352294922,
"InsertCoreBarrier": 0.1479203701019287,
"InsertIOTransposes": 2.3931386470794678,
"InsertImplicitShardAxisBeforeISel": 0.5525217056274414,
"InsertLocalTransposes": 0.7429313659667969,
"InsertOffloadedTransposes": 0.19022369384765625,
"LICM": 0.14770793914794922,
"LateLegalizeInst": 0.30222225189208984,
"LateLegalizePostSplit": 0.22165155410766602,
"LateLowerReshapeOp": 0.03522944450378418,
"LateLowerTensorOp": 6.813569068908691,
"LateNeuronInstComb": 1.2631995677947998,
"LayoutPreprocessing": 1.8736376762390137,
"LayoutPreprocessingAndAnalysis": 2.3256783485412598,
"LayoutRequirementAnalysis": 0.4448714256286621,
"LegalizeCCOpLayout": 1.317772388458252,
"LegalizeOpLevelAlias": 0.5106728076934814,
"LegalizePartitionReduce": 0.08073091506958008,
"LegalizeSundaAccess": 0.9303104877471924,
"LegalizeSundaMacro": 0.7451527118682861,
"LegalizeType": 0.12957215309143066,
"LocalLayoutOpt": 0.6608095169067383,
"LoopFusion": 0.3101928234100342,
"LoopSplitting": 0.036502838134765625,
"LowerBroadcast": 0.06164741516113281,
"LowerCCOpBlockAxis": 0.32977890968322754,
"LowerComplexBroadcast": 0.15041875839233398,
"LowerIntrinsics": 1.0597736835479736,
"LowerShardAxis": 0.4116206169128418,
"LowerTensorOp": 3.285423755645752,
"LowerToSendRecv": 0.25904369354248047,
"LowerTranspose": 0.5202181339263916,
"MacroGeneration": 2.1192569732666016,
"MaskPropagation": 0.14647436141967773,
"MemcpyElimination": 29.3009090423584,
"MutateDataType": 0.037621498107910156,
"NeuronAliasDependencyInduction": 0.06313967704772949,
"NeuronAliasDependencyReset": 0.08403873443603516,
"NeuronInstComb": 0.3526570796966553,
"NeuronLICM": 0.28780031204223633,
"NeuronLoopFusion": 1.491703987121582,
"NeuronLoopInterchange": 0.06900668144226074,
"NeuronSimplifier": 0.5315425395965576,
"NeuronSimplifyPredicates": 0.25080394744873047,
"NeuronValueNumbering": 0.11057853698730469,
"OptimizeAliasedCopyChain": 0.6118557453155518,
"OptimizeNKIKernels": 1.0786433219909668,
"PAGLayoutOpt": 18.570398330688477,
"PComputeCutting": 0.6222004890441895,
"PGLayoutTilingPipeline": 55.62285614013672,
"PGTiling": 10.0595703125,
"PadElimination": 0.021188735961914063,
"ParAxesAnnotation": 17.818796157836914,
"PartialLoopFusion": 1.5088038444519043,
"PartialSimdFusion": 0.8573262691497803,
"PerfectLoopNest": 0.07143402099609375,
"RecognizeOpIdiom": 0.1291654109954834,
"Recompute": 0.007833480834960938,
"RelaxPredicates": 0.14508056640625,
"Rematerialization": 0.17157793045043945,
"RemoveShardedPartitionAxes": 1.3566563129425049,
"ReshapeWeights": 0.022662878036499023,
"ResolveAccessConflict": 0.2033238410949707,
"ResolveComplicatePredicates": 0.49439358711242676,
"RewriteReplicationMatmul": 0.03938794136047363,
"RewriteWeights": 0.07809877395629883,
"SFKVectorizer": 10.313403129577637,
"ShardingPropagationAnalysis": 0.7199399471282959,
"SimpleAllReduceTiling": 0.06984925270080566,
"Simplifier": 0.09367704391479492,
"SimplifyMacroPredicates": 0.28262877464294434,
"SimplifyNeuronTensor": 0.9048190116882324,
"SimplifySlice": 0.027447223663330078,
"SimplifyTensor": 0.2734675407409668,
"SpillPSum": 1.2256271839141846,
"SplitAPUnionSets": 0.8627588748931885,
"SplitAccGrp": 0.1022341251373291,
"StaticProfiler": 0.4753732681274414,
"StaticTransposeLocalTensor": 0.6612956523895264,
"SundaISel": 1.6735970973968506,
"TCTransform": 0.029105186462402344,
"TensorInitialization": 0.1944262981414795,
"TensorOpSimplifier": 3.3376035690307617,
"TensorOpTransform": 19.926769256591797,
"TileCCOps": 0.1856687068939209,
"TilingProfiler": 0.5146362781524658,
"TransformConvOp": 1.4806337356567383,
"TritiumFusion": 0.2828052043914795,
"ValueNumbering": 0.08587479591369629,
"VectorizeDMA": 0.5646088123321533,
"VectorizeMatMult": 0.08723974227905273,
"WeightCoalescing": 0.06122708320617676,
"ZeroSizeTensorElimination": 0.0008580684661865234
},
"tensorizer": {
"DMATilingProfiler::TotalInstructionsAfterTiling": 51674,
"StaticProfiler::AifUb": 21.735599517822266,
"StaticProfiler::ArithmeticIntensityTensorizer": 31.884082794189453,
"StaticProfiler::AverageDmaLength": 2843.9130859375,
"StaticProfiler::AverageFractalPeUtilization": 97.88961791992188,
"StaticProfiler::AveragePartitionUtilization": 89.79463958740234,
"StaticProfiler::AveragePeUtilization": 78.904052734375,
"StaticProfiler::DDRTransferBytes": 1013018224,
"StaticProfiler::InternalTransferBytes": 226623072,
"StaticProfiler::LoadExpanded": 281500,
"StaticProfiler::LocalizationEfficiency": 146.6905975341797,
"StaticProfiler::LocalizationEfficiencyIgnoreNonlocal": 151.06674194335938,
"StaticProfiler::StoreExpanded": 4273,
"StaticProfiler::TotalDMAExpanded": 285773,
"StaticProfiler::TotalDynamicInstancesCount": 67989,
"StaticProfiler::TotalDynamicInstancesWithMmPackedCount": 57540,
"StaticProfiler::TotalLNCComm": 0,
"StaticProfiler::TotalLNCCommTransfer": 0,
"TilingProfiler::AveragePartitionUtilizationAfterTiling": 0,
"TilingProfiler::AveragePeUtilizationAfterTiling": 0,
"TilingProfiler::BatchnormInstructionsAfterTiling": 0,
"TilingProfiler::DmaInstructionsAfterTiling": 0,
"TilingProfiler::GenericInstructionsAfterTiling": 123,
"TilingProfiler::MatMultInstructionsAfterTiling": 35008,
"TilingProfiler::NumPfTransposes": 292,
"TilingProfiler::NumPfTransposesForIo": 30,
"TilingProfiler::NumPfTransposesForLocal": 142,
"TilingProfiler::NumPfTransposesForNonlocal": 120,
"TilingProfiler::PfTransposeInstructions": 10670,
"TilingProfiler::PfTransposeInstructionsForIo": 8360,
"TilingProfiler::PfTransposeInstructionsForLocal": 648,
"TilingProfiler::PfTransposeInstructionsForNonlocal": 1662,
"TilingProfiler::ReduceInstructionsAfterTiling": 283,
"TilingProfiler::SimdInstructionsAfterTiling": 2805,
"TilingProfiler::TotalInstructionsAfterTiling": 0,
"TransformConvOp::Conv1d_depthwise_bf01_oi01_bf01": 0,
"TransformConvOp::Conv2d_dw_fb01_io01_01bf_rep_nhwc_Pcinh": 0,
"TransformConvOp::Conv2d_pbp_0f1b_0i1o_01fb_experimental_1": 0,
"TransformConvOp::Conv2d_pbp_fb01_io01_01bf_experimental_1": 0,
"TransformConvOp::conv2d_column_packing": 0,
"TransformConvOp::conv2d_column_packing_1": 0,
"TransformConvOp::conv2d_column_packing_io10": 0,
"TransformConvOp::conv2d_depthwise_f01b_o01i_bf01": 0
}
},
"topk": {
"compiletime": {
"CoalesceCCOp": 0.003665924072265625,
"DMALocalityOpt": 0.002828359603881836,
"DMAProfiler": 0.0035741329193115234,
"DataStreaming": 0.006661653518676758,
"DoNothing": 0.00015878677368164063,
"ExpandISAMacro": 0.0038259029388427734,
"FactorizeBlkDims": 0.012358903884887695,
"InferPSumTensor": 0.01241612434387207,
"InferSharedMemLoc": 0.0028722286224365234,
"InsertCoreBarrier": 0.0033416748046875,
"LateLegalizeInst": 0.0075190067291259766,
"LateNeuronInstComb": 0.008240699768066406,
"LegalizeSundaAccess": 0.014475584030151367,
"LegalizeType": 0.00968170166015625,
"LowerBroadcast": 0.0032701492309570313,
"LowerIntrinsics": 0.0037560462951660156,
"LowerTranspose": 0.0033812522888183594,
"NeuronInstComb": 0.008539199829101563,
"NeuronLICM": 0.00936579704284668,
"NeuronSimplifyPredicates": 0.0037877559661865234,
"NeuronValueNumbering": 0.0040531158447265625,
"SFKVectorizer": 0.03392672538757324,
"SimpleAllReduceTiling": 0.003645181655883789,
"SimplifyNeuronTensor": 0.05141305923461914,
"SpillPSum": 0.02246880531311035,
"WeightCoalescing": 0.0034682750701904297
}
}
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0c81d96024b571f29492feb3e8f54a86bb85699475d770c8027b26b3b25d6d30
size 3605504

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6f9340d294060545b7fd3affe72e9160857b035623ad5e43d5186704d48fc4cd
size 2464555

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:25287d0dc180171fa1e47e9880d832a32168762120f79d5891ed820ee9b4fcda
size 2550451

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0c81d96024b571f29492feb3e8f54a86bb85699475d770c8027b26b3b25d6d30
size 3605504

View File

@@ -0,0 +1,224 @@
{
"_attn_implementation_autoset": false,
"_name_or_path": "/models/Qwen3-1.7B/",
"add_cross_attention": false,
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attribute_map": {},
"bad_words_ids": null,
"begin_suppress_tokens": null,
"bos_token_id": 151643,
"chunk_size_feed_forward": 0,
"cross_attention_hidden_size": null,
"decoder_start_token_id": null,
"diversity_penalty": 0.0,
"do_sample": false,
"early_stopping": false,
"encoder_no_repeat_ngram_size": 0,
"eos_token_id": 151645,
"exponential_decay_length_penalty": null,
"finetuning_task": null,
"forced_bos_token_id": null,
"forced_eos_token_id": null,
"fused_spec_config": null,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"initializer_range": 0.02,
"intermediate_size": 6144,
"is_decoder": false,
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"length_penalty": 1.0,
"max_length": 20,
"max_position_embeddings": 40960,
"max_window_layers": 28,
"metadata": null,
"min_length": 0,
"model_type": "qwen3",
"neuron_config": {
"activation_quantization_type": null,
"allow_input_truncation": false,
"apply_seq_ids_mask": false,
"async_mode": false,
"attention_dp_degree": 1,
"attention_dtype": null,
"attn_block_cte_nki_kernel_enabled": false,
"attn_block_tkg_nki_kernel_cache_update": false,
"attn_block_tkg_nki_kernel_cascaded_attention": false,
"attn_block_tkg_nki_kernel_enabled": false,
"attn_cls": {
"__module__": "neuronx_distributed_inference.models.qwen3.modeling_qwen3",
"__name__": "NeuronQwen3Attention"
},
"attn_kernel_enabled": null,
"attn_tkg_builtin_kernel_enabled": false,
"attn_tkg_nki_kernel_enabled": false,
"batch_size": 4,
"bucket_n_active_tokens": false,
"buckets": [
1024
],
"cast_type": "config",
"cc_pipeline_tiling_factor": 1,
"chunked_prefill_config": null,
"context_encoding_buckets": null,
"cp_degree": 1,
"ctx_batch_size": 1,
"disable_kv_cache_tiling": false,
"draft_model_modules_to_not_convert": null,
"enable_bucketing": true,
"enable_cte_modular_flow": false,
"enable_eagle_draft_input_norm": false,
"enable_eagle_speculation": false,
"enable_fused_speculation": false,
"enable_long_context_mode": false,
"enable_output_completion_notifications": false,
"enable_spill_reload_dge": false,
"enable_token_tree": false,
"ep_degree": 1,
"expert_mlp_nki_kernel_enabled": null,
"flash_decoding_enabled": false,
"fused_qkv": false,
"fused_rmsnorm_skip_gamma": false,
"is_block_kv_layout": null,
"is_chunked_prefill": false,
"is_continuous_batching": true,
"is_eagle_draft": false,
"is_medusa": false,
"is_prefill_stage": false,
"is_prefix_caching": false,
"k_cache_transposed": false,
"kv_cache_batch_size": 4,
"kv_cache_padding_size": 0,
"kv_cache_quant": false,
"kv_cache_tiling": false,
"layer_boundary_markers": false,
"lm_head_pad": true,
"lm_head_pad_alignment_size": 1,
"local_ranks_size": 4,
"logical_nc_config": 2,
"lora_config": null,
"max_batch_size": 4,
"max_context_length": 2048,
"max_length": 2048,
"max_new_tokens": null,
"medusa_speculation_length": 0,
"medusa_tree": null,
"mlp_kernel_enabled": false,
"mlp_kernel_fuse_residual_add": false,
"modules_to_not_convert": null,
"moe_fused_nki_kernel_enabled": null,
"n_active_tokens": 1,
"n_positions": 2048,
"num_medusa_heads": 0,
"on_cpu": false,
"on_device_sampling_config": {
"deterministic": false,
"do_sample": false,
"dynamic": true,
"global_topk": 256,
"on_device_sampling_config": true,
"temperature": 1.0,
"top_k": 1,
"top_k_kernel_enabled": false,
"top_p": 1.0
},
"output_logits": false,
"overrides_torch_dtype": true,
"pa_block_size": 2048,
"pa_num_blocks": 4,
"padding_side": "right",
"pp_degree": 1,
"prefix_buckets": null,
"qk_layernorm": false,
"qkv_kernel_enabled": false,
"qkv_kernel_fuse_residual_add": false,
"qkv_kernel_nbsd_layout": false,
"quantization_dtype": "int8",
"quantization_type": "per_tensor_symmetric",
"quantize_clamp_bound": Infinity,
"quantized": false,
"quantized_checkpoints_path": null,
"quantized_mlp_kernel_enabled": false,
"rmsnorm_quantize_kernel_enabled": false,
"router_topk_nki_kernel_enabled": null,
"rpl_reduce_dtype": null,
"save_sharded_checkpoint": true,
"scratchpad_page_size": null,
"seq_len": 2048,
"seq_len_threshold_for_cc_tiling": 16384,
"sequence_parallel_enabled": false,
"shared_mlp_nki_kernel_enabled": null,
"skip_sharding": false,
"skip_warmup": false,
"spec_batch_size": 4,
"speculation_length": 0,
"start_rank_id": 0,
"strided_context_parallel_kernel_enabled": false,
"target": null,
"tensor_capture_config": null,
"tile_cc": false,
"tkg_batch_size": 4,
"token_generation_buckets": [
1024
],
"token_tree_config": null,
"torch_dtype": "bfloat16",
"tp_degree": 4,
"vocab_parallel": false,
"weight_gather_seq_len_threshold": 32768,
"weights_to_skip_layout_optimization": [],
"world_size": 4
},
"no_repeat_ngram_size": 0,
"num_attention_heads": 16,
"num_beam_groups": 1,
"num_beams": 1,
"num_cores_per_group": 1,
"num_hidden_layers": 28,
"num_key_value_heads": 8,
"num_return_sequences": 1,
"output_attentions": false,
"output_hidden_states": false,
"output_scores": false,
"pad_token_id": 0,
"prefix": null,
"problem_type": null,
"pruned_heads": {},
"remove_invalid_values": false,
"repetition_penalty": 1.0,
"return_dict": true,
"return_dict_in_generate": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000,
"sep_token_id": null,
"sliding_window": null,
"suppress_tokens": null,
"task_specific_params": null,
"temperature": 1.0,
"tf_legacy_loss": false,
"tie_encoder_decoder": false,
"tie_word_embeddings": true,
"tokenizer_class": null,
"top_k": 50,
"top_p": 1.0,
"torchscript": false,
"transformers_version": "4.51.0",
"typical_p": 1.0,
"use_bfloat16": false,
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1 @@
neuronx-cc compile --framework=XLA model.MODULE_b4b44a47076a167bd9a5+795ee7cc.hlo_module.pb --output model.MODULE_b4b44a47076a167bd9a5+795ee7cc.neff --target=trn2 --auto-cast=none --model-type=transformer '--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ' --lnc=2 -O2 --internal-hlo2tensorizer-options=--verify-hlo=true --logfile=log-neuron-cc.txt --verbose=35

View File

@@ -0,0 +1 @@
["--target=trn2", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=1 --vectorize-strided-dma ", "--lnc=2", "-O2", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/models/Qwen3-1.7B-compiled/token_generation_model/_tp0_bk4/log-neuron-cc.txt"]

Some files were not shown because too many files have changed in this diff Show More