GitHub
182.5K
Download
Models
Docs
Pricing
Sign in
Download
Models
Docs
Pricing
Sign in
jetelain
/
Gemma-4-31B
:latest
172
Downloads
Updated
3 months ago
Gemma-4-31B UD-Q4_K_XL MTP from Unsloth
Gemma-4-31B UD-Q4_K_XL MTP from Unsloth
Cancel
vision
tools
thinking
Gemma-4-31B:latest
...
/
draft
5ae8b0117bed · 515MB
Metadata
general.architecture
gemma4-assistant
gemma4-assistant
general.file_type
Q8_0
Q8_0
gemma4-assistant.attention.head_count
32
32
gemma4-assistant.attention.head_count_kv
[16, 16, 16, 4]
[16, 16, 16, 4]
gemma4-assistant.attention.key_length
512
512
gemma4-assistant.attention.key_length_swa
256
256
gemma4-assistant.attention.layer_norm_rms_epsilon
1e-06
1e-06
gemma4-assistant.attention.shared_kv_layers
4
4
gemma4-assistant.attention.sliding_window
1024
1024
gemma4-assistant.attention.sliding_window_pattern
[true, true, true, false]
[true, true, true, false]
gemma4-assistant.attention.value_length
512
512
gemma4-assistant.attention.value_length_swa
256
256
gemma4-assistant.block_count
4
4
gemma4-assistant.context_length
262144
262144
gemma4-assistant.embedding_length
1024
1024
gemma4-assistant.embedding_length_out
5376
5376
gemma4-assistant.embedding_length_per_layer_input
0
0
gemma4-assistant.feed_forward_length
8192
8192
gemma4-assistant.nextn_predict_layers
4
4
gemma4-assistant.rope.dimension_count
512
512
gemma4-assistant.rope.dimension_count_swa
256
256
gemma4-assistant.rope.freq_base
1e+06
1e+06
gemma4-assistant.rope.freq_base_swa
10000
10000
tokenizer.ggml.add_bos_token
true
true
tokenizer.ggml.add_space_prefix
false
false
tokenizer.ggml.bos_token_id
2
2
tokenizer.ggml.eos_token_id
1
1
tokenizer.ggml.mask_token_id
4
4
tokenizer.ggml.merges
[ , ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ ▁, , , ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ ▁▁, ...]
[ , ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ ▁, , , ▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁▁ ▁▁, ...]
tokenizer.ggml.model
gemma4
gemma4
tokenizer.ggml.padding_token_id
0
0
tokenizer.ggml.scores
[-1000, -1000, -1000, -1000, -1000, ...]
[-1000, -1000, -1000, -1000, -1000, ...]
tokenizer.ggml.token_type
[3, 3, 3, 3, 3, ...]
[3, 3, 3, 3, 3, ...]
tokenizer.ggml.tokens
[<pad>, <eos>, <bos>, <unk>, <mask>, ...]
[<pad>, <eos>, <bos>, <unk>, <mask>, ...]
tokenizer.ggml.unknown_token_id
3
3
Tensor
Name
Type
Shape
token_embd.weight
Q8_0
Q8_0
[1024, 262144]
blk.0
blk.0.attn_norm.weight
F32
F32
[1024]
blk.0.attn_output.weight
Q8_0
Q8_0
[8192, 1024]
blk.0.attn_q.weight
Q8_0
Q8_0
[1024, 8192]
blk.0.attn_q_norm.weight
F32
F32
[256]
blk.0.ffn_down.weight
Q8_0
Q8_0
[8192, 1024]
blk.0.ffn_gate.weight
Q8_0
Q8_0
[1024, 8192]
blk.0.ffn_norm.weight
F32
F32
[1024]
blk.0.ffn_up.weight
Q8_0
Q8_0
[1024, 8192]
blk.0.layer_output_scale.weight
F32
F32
[1]
blk.0.post_attention_norm.weight
F32
F32
[1024]
blk.0.post_ffw_norm.weight
F32
F32
[1024]
blk.1
blk.1.attn_norm.weight
F32
F32
[1024]
blk.1.attn_output.weight
Q8_0
Q8_0
[8192, 1024]
blk.1.attn_q.weight
Q8_0
Q8_0
[1024, 8192]
blk.1.attn_q_norm.weight
F32
F32
[256]
blk.1.ffn_down.weight
Q8_0
Q8_0
[8192, 1024]
blk.1.ffn_gate.weight
Q8_0
Q8_0
[1024, 8192]
blk.1.ffn_norm.weight
F32
F32
[1024]
blk.1.ffn_up.weight
Q8_0
Q8_0
[1024, 8192]
blk.1.layer_output_scale.weight
F32
F32
[1]
blk.1.post_attention_norm.weight
F32
F32
[1024]
blk.1.post_ffw_norm.weight
F32
F32
[1024]
blk.2
blk.2.attn_norm.weight
F32
F32
[1024]
blk.2.attn_output.weight
Q8_0
Q8_0
[8192, 1024]
blk.2.attn_q.weight
Q8_0
Q8_0
[1024, 8192]
blk.2.attn_q_norm.weight
F32
F32
[256]
blk.2.ffn_down.weight
Q8_0
Q8_0
[8192, 1024]
blk.2.ffn_gate.weight
Q8_0
Q8_0
[1024, 8192]
blk.2.ffn_norm.weight
F32
F32
[1024]
blk.2.ffn_up.weight
Q8_0
Q8_0
[1024, 8192]
blk.2.layer_output_scale.weight
F32
F32
[1]
blk.2.post_attention_norm.weight
F32
F32
[1024]
blk.2.post_ffw_norm.weight
F32
F32
[1024]
blk.3
blk.3.attn_norm.weight
F32
F32
[1024]
blk.3.attn_output.weight
Q8_0
Q8_0
[16384, 1024]
blk.3.attn_q.weight
Q8_0
Q8_0
[1024, 16384]
blk.3.attn_q_norm.weight
F32
F32
[512]
blk.3.ffn_down.weight
Q8_0
Q8_0
[8192, 1024]
blk.3.ffn_gate.weight
Q8_0
Q8_0
[1024, 8192]
blk.3.ffn_norm.weight
F32
F32
[1024]
blk.3.ffn_up.weight
Q8_0
Q8_0
[1024, 8192]
blk.3.layer_output_scale.weight
F32
F32
[1]
blk.3.post_attention_norm.weight
F32
F32
[1024]
blk.3.post_ffw_norm.weight
F32
F32
[1024]
nextn.post_projection.weight
Q8_0
Q8_0
[1024, 5376]
nextn.pre_projection.weight
Q8_0
Q8_0
[10752, 1024]
rope_freqs.weight
F32
F32
[256]
output_norm.weight
F32
F32
[1024]