50 lines
1.1 KiB
YAML
50 lines
1.1 KiB
YAML
kind: nla_model
|
|
schema_version: 2
|
|
d_model: 1536
|
|
tokens:
|
|
injection_char: <|image_pad|>
|
|
injection_token_id: 151655
|
|
injection_left_neighbor_id: 220
|
|
injection_right_neighbor_id: 151645
|
|
critic_suffix_ids: null
|
|
prompt_templates:
|
|
av: '<|im_start|>system
|
|
|
|
You interpret neural-network activations. Given one activation vector, name in a single sentence the
|
|
concept, entity, topic, or syntactic role it encodes. Be concrete; do not hedge or add preamble.<|im_end|>
|
|
|
|
<|im_start|>user
|
|
|
|
Activation: {injection_char}<|im_end|>
|
|
|
|
<|im_start|>assistant
|
|
|
|
'
|
|
ar: '{explanation}'
|
|
created_by: nla (dormantx/NLA_Qwen2.5_1.5B pipeline)
|
|
role_aliases:
|
|
verbalizer: actor
|
|
recon: critic
|
|
layer: 18
|
|
extraction_layer_index: 18
|
|
base_model: Qwen/Qwen2.5-1.5B-Instruct
|
|
role: ar
|
|
stage: sl
|
|
extraction:
|
|
injection_scale: null
|
|
mse_scale: 1.0
|
|
critic:
|
|
extraction_layer_index: 18
|
|
training:
|
|
lr: 5.0e-05
|
|
loss_type: mse_unit_norm
|
|
global_batch_size: 32
|
|
num_layers: 18
|
|
value_head:
|
|
file: value_head.safetensors
|
|
dtype: float32
|
|
weight_shape:
|
|
- 1536
|
|
- 1536
|
|
bias: true
|