Team Ai
Modelpublic

parkingsoman/gemma-4-e2b-it-onnx-webgpu

sourceHugging Faceapache-2.0updated 13d agoView on Hugging Face
0likes32downloads
Model Card

Gemma 4 E2B IT — ONNX for WebGPU in the browser

This is the "Small" model package behind Axiya. Axiya turns scanned PDFs into accessible HTML. The model runs inside Chrome, on the user's own GPU. No document leaves the computer.

The weights come from Google's Gemma 4 E2B IT QAT release, repacked for ONNX Runtime Web. The graphs need the Axiya onnxruntime-web fork (op SplitMaskedSoftmax, fp32 accumulation). The download is about 3.1 GB.

The build report below lists every input with its revision or sha256.

Gemma 4 E2B Q4 ONNX package (browser, ORT Web)

License: Apache 2.0 (Gemma 4 terms: https://ai.google.dev/gemma/docs/gemma4license). The graphs are our own export with MIT tools (tools/gemma12b/mobius_export.py); the weights are Google's.

Rebuild: browser-prototype/tools/gemma12b/README.md in the BlinkFix repo.

Sources

InputSourceRevision / sha256
Graph structuremobius (onnxruntime/mobius 1c2dc4d, MIT) + Olive 0.13.0 k-quant (MIT), from google/gemma-4-E2B-it 3e22461 (graph files only)see below
graph decoder/model.onnxBlinkFix-models/mobius-e2b/q49681ede78ae69b4a16bef2dc1a0833a69465545136ed6977436013f442c511f1 (checked)
graph embedding/model.onnxBlinkFix-models/mobius-e2b/q4cc9bb5218d38455681b90dd384038af072599245fab94028d1a1acbd8c727270 (checked)
graph vision_encoder/model.onnxBlinkFix-models/mobius-e2b/q40f2c5c42a65723d9dc56c224e4cd599079dc0fe09b3b14efad617145099a841e (checked)
Weights e2b-ud-plmp5365.gguflocal file0edce72e2ea1b43bb0d9864038a87172d91ded391431f992ee6cc427351d035a
Weights gemma-4-E2B-it-mmproj.gguflocal file021059cce659fe7f9170d5599761d7bbaf644b798dab9503aca30dc43e6beb14
Tokenizer, chat template, configsgoogle/gemma-4-E2B-it3e22461f65e89153144f8adb70e3b8c2cc9845a7
Norm-convention and layout checks (range reads)google/gemma-4-E2B-it-qat-q4_0-unquantized6befbaca7398925921802abd1f277b495b78b738
PLE projection (F16, x53.65)google/gemma-4-E2B-it-qat-q40-unquantized@6befbaca7398925921802abd1f277b495b78b738 model.languagemodel.perlayermodel_projection.weight (range read)bf16 bf07beb16867b9e3ea03cac3af1650dfdd5448bbde999f8265b235cea5205e26

Graphs

GraphInputsOutputs
vision.onnxpixel_values fp16 [1,N,768] (HF Gemma4ImageProcessor patches, 16 px, no padding rows), pixel_position_ids int64 [1,N,2] (x,y)image_features fp16 [N/9,1536]
decoder.onnxinput_ids int64 [B,S], image_features fp16 [N,1536], ple_rows float32 [B,S,8960] (see ple.json), attention_mask int64 [B,T], position_ids int64 [B,S], num_logits_to_keep int64 scalar. The embedding lookup reads the 4-bit lmhead table (one shared copy). KV cache for layers 0..14 only (layers 15..34 read layer 13 or 14), split layout: stable `pastkeyvalues.{0..14}.key` fp16 [B,kv,d,P] (transposed), `.value` [B,kv,P,d]; recent `pastrecent.{0..14}.*`logits fp16 [B,num_logits,262144], present.{0..14}.key [B,kv,d,R+S], .value [B,kv,R+S,d]
kvmerge.onnx (no weights)the decoder's stable and recent cache inputsmerged.{0..14}.*: the next stable part
ple.q4_0_*.bin, ple.jsonnot a graph: the per-layer table's raw Q4_0 rows, read per token by the engine

Files

FileBytessha256
chat_template.jinja185690a2c8073c878ab1da004bee933a998606537bbb62016310352c7285c3f01c5b5
config.json49541b28f3d2c3100f6c594754b81107428bd7b822a7f48272ca681dae9d2ec38330
decoder.h0.onnx943374273c9fcbb12ace71b322edf3051a209c1c8a0aac64060750440af9fb6ca90894
decoder.onnx9380602ad469e704dbf5c848f0db1f3e877e117c291064067380e01d22b19d744e4bed
decoder.onnx_data13278740484e8bbc38f9e1823d0d0686788a2189e237529fa275b60a5b402fabe7bb08bc1b
drafter.json469175d015b589be05c7391d6ee84221315be0e780acc6a752ceb2e626a0f8672d0
drafter.onnx290646663801f62847836417cf31496939b04122c8e30f7946ab89174962b26bea6d7
drafter.onnx_data6859264058b89543fa5ea6fbea6c2a5b285710ffa8094126132f40121e1acdff35e9080a
generation_config.json208d4226bbe3117d2d253ba4609720ba82c6c4ce4627a9a6ae05387c78983ac03de
kvmerge.onnx9140a7f15f62a4a86cd7e86d15500c9fb881776f54e22cca023b03f14852446c64a0
ple.json110322c82333e6de06989fcc5cbbe27f656210cd4562fd01cac2140c5370ce73f6e7
ple.q4_0_0.bin13212057605ebf6a486f7e55fed216d3706adf66f8c98ce066e8430b450e215625fd0ad67a
processor_config.json168932bdf45d2ad4cc29a0822ddd157a182de76644f0419a6228d151495256e9813c
tokenizer.json32169626cc8d3a0ce36466ccc1278bf987df5f71db1719b9ca6b4118264f45cb627bfe0f
tokenizer_config.json30829f4fec4b1dc6ecddf8f4a92e9caea5971c0e67d81309f3f9066a2bee8c362633
vision.onnx1246318a229fed8eebd4c7a1ba74eb203a5768330c76cd0f232b566ec2c7f5fad9ee514
vision.onnx_data358222848e4e276760b1f1a20629d5470e54f940a59813600f50cd2ea90bed2a58ee27dd5

Build log

json
{
 "graph_dir": "BlinkFix-models/mobius-e2b/q4",
 "graph_sha256": {
  "decoder/model.onnx": "9681ede78ae69b4a16bef2dc1a0833a69465545136ed6977436013f442c511f1",
  "embedding/model.onnx": "cc9bb5218d38455681b90dd384038af072599245fab94028d1a1acbd8c727270",
  "vision_encoder/model.onnx": "0f2c5c42a65723d9dc56c224e4cd599079dc0fe09b3b14efad617145099a841e"
 },
 "qk_precision": "fp32",
 "decoder": {
  "causal_proof": {
   "v_decoder.model.Unsqueeze_110": {
    "causal": true,
    "pad_blocked": true,
    "window": null,
    "min_blocked_dist": null,
    "max_allowed_dist": 1025
   },
   "v_decoder.model.Unsqueeze_96": {
    "causal": true,
    "pad_blocked": true,
    "window": 512,
    "min_blocked_dist": 512,
    "max_allowed_dist": 511
   }
  },
  "attention": [
   {
    "name": "decoder/model/layers.0/self_attn/Attention_node_137",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.0.key",
    "past_value": "past_key_values.0.value",
    "present_key": "present.0.key",
    "present_value": "present.0.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.1/self_attn/Attention_node_188",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.1.key",
    "past_value": "past_key_values.1.value",
    "present_key": "present.1.key",
    "present_value": "present.1.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.2/self_attn/Attention_node_239",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.2.key",
    "past_value": "past_key_values.2.value",
    "present_key": "present.2.key",
    "present_value": "present.2.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.3/self_attn/Attention_node_290",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.3.key",
    "past_value": "past_key_values.3.value",
    "present_key": "present.3.key",
    "present_value": "present.3.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.4/self_attn/Attention_node_341",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "past_key": "past_key_values.4.key",
    "past_value": "past_key_values.4.value",
    "present_key": "present.4.key",
    "present_value": "present.4.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.5/self_attn/Attention_node_392",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.5.key",
    "past_value": "past_key_values.5.value",
    "present_key": "present.5.key",
    "present_value": "present.5.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.6/self_attn/Attention_node_443",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.6.key",
    "past_value": "past_key_values.6.value",
    "present_key": "present.6.key",
    "present_value": "present.6.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.7/self_attn/Attention_node_494",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.7.key",
    "past_value": "past_key_values.7.value",
    "present_key": "present.7.key",
    "present_value": "present.7.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.8/self_attn/Attention_node_545",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.8.key",
    "past_value": "past_key_values.8.value",
    "present_key": "present.8.key",
    "present_value": "present.8.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.9/self_attn/Attention_node_596",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "past_key": "past_key_values.9.key",
    "past_value": "past_key_values.9.value",
    "present_key": "present.9.key",
    "present_value": "present.9.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.10/self_attn/Attention_node_647",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.10.key",
    "past_value": "past_key_values.10.value",
    "present_key": "present.10.key",
    "present_value": "present.10.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.11/self_attn/Attention_node_698",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.11.key",
    "past_value": "past_key_values.11.value",
    "present_key": "present.11.key",
    "present_value": "present.11.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.12/self_attn/Attention_node_749",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.12.key",
    "past_value": "past_key_values.12.value",
    "present_key": "present.12.key",
    "present_value": "present.12.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.13/self_attn/Attention_node_800",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "past_key": "past_key_values.13.key",
    "past_value": "past_key_values.13.value",
    "present_key": "present.13.key",
    "present_value": "present.13.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.14/self_attn/Attention_node_851",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "past_key": "past_key_values.14.key",
    "past_value": "past_key_values.14.value",
    "present_key": "present.14.key",
    "present_value": "present.14.value",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.15/self_attn/Attention_node_887",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.16/self_attn/Attention_node_923",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.17/self_attn/Attention_node_959",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.18/self_attn/Attention_node_995",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.19/self_attn/Attention_node_1031",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.14/self_attn/Attention_node_851",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.20/self_attn/Attention_node_1067",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.21/self_attn/Attention_node_1103",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.22/self_attn/Attention_node_1139",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.23/self_attn/Attention_node_1175",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.24/self_attn/Attention_node_1211",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.14/self_attn/Attention_node_851",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.25/self_attn/Attention_node_1247",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.26/self_attn/Attention_node_1283",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.27/self_attn/Attention_node_1319",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.28/self_attn/Attention_node_1355",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.29/self_attn/Attention_node_1391",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.14/self_attn/Attention_node_851",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.30/self_attn/Attention_node_1427",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.31/self_attn/Attention_node_1463",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.32/self_attn/Attention_node_1499",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.33/self_attn/Attention_node_1535",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 256,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.13/self_attn/Attention_node_800",
    "is_causal": 1
   },
   {
    "name": "decoder/model/layers.34/self_attn/Attention_node_1571",
    "q_heads": 8,
    "kv_heads": 1,
    "head_dim": 512,
    "scale": 1.0,
    "shared_from": "decoder/model/layers.14/self_attn/Attention_node_851",
    "is_causal": 1
   }
  ],
  "kv_layout": "split",
  "or_replaced": 0,
  "mask_chain_on_cpu": [],
  "mask_gate_no_bool_broadcast": [
   "decoder/model/Where_node_94",
   "decoder/model/Where_node_108"
  ],
  "split_softmax_fused": 35,
  "sliding_kv_trim": {
   "sliding_layers": 28,
   "mask_users": {
    "SplitMaskedSoftmax": 28
   },
   "replaced_tile": "v_decoder.model.Unsqueeze_96/tile8",
   "mask_source": "v_decoder.model.Unsqueeze_96/f32"
  },
  "zero_points_dropped": 276,
  "sliced_hidden": "v_decoder.model.layers.34.Mul_1595",
  "equal": [],
  "table_bits": 4,
  "decoder_files": [
   "decoder.onnx_data"
  ],
  "decoder_seconds": 1.9,
  "seconds": 9.4
 },
 "embed": {
  "ple_rows_input": {
   "table_removed": "embedding.embed_tokens_per_layer.weight",
   "input": "ple_rows",
   "width": 8960,
   "scale": 16.0,
   "replaced": [
    "embedding/embed_tokens_per_layer/Gather_node_47",
    "embedding/embed_tokens_per_layer/Mul_node_48"
   ]
  },
  "mask_chain_on_cpu": [
   "embedding/Cast_node_5"
  ],
  "embed_equal": [
   {
    "node": "embedding/Equal_node_3",
    "inputs": [
     "input_ids",
     "v_embedding.Constant_2"
    ],
    "upstream_graph_inputs": [
     "input_ids"
    ],
    "upstream_ops": [
     "Constant"
    ],
    "action": "left as is"
   }
  ],
  "ple_projection": {
   "source": "google/gemma-4-E2B-it-qat-q4_0-unquantized@6befbaca7398925921802abd1f277b495b78b738 model.language_model.per_layer_model_projection.weight",
   "bf16_sha256": "bf07beb16867b9e3ea03cac3af1650dfdd5448bbde999f8265b235cea5205e26",
   "scale": 53.65,
   "dtype": "float16 (bf16 x scale, rounded once)"
  },
  "embed_files": [
   "embed.onnx_data"
  ],
  "seconds": 7.2
 },
 "vision": {
  "vision_files": [
   "vision.onnx_data"
  ],
  "vision_mmproj_tensors": 659,
  "vision_inline_checked": 480,
  "vision_layers": 16,
  "vision_dtype": [
   "float16",
   "float32"
  ],
  "seconds": 0.3
 },
 "ple": {
  "ple": {
   "files": [
    "ple.q4_0_0.bin"
   ],
   "bytes": 1321205760,
   "rows": 262144
  },
  "seconds": 9.7
 },
 "shared_embed": {
  "table_sha256": {
   "decoder.lm_head.weight_t_Q4": "d944107936c0be4ca7a6fe81ea2d88845499dda208a89434efc966e1314a74ba",
   "decoder.lm_head.weight_t_scales": "57748db985c30e8c5a448228ab05311f7144b4cc715468fef3d36d55b93e3b78"
  },
  "embed_nodes": 27,
  "added_initializers": [
   "decoder.lm_head.weight_t_scales.gather",
   "const_39.191835884530846_f16",
   "g12_embed/const_1d_1",
   "const_0.02551551815399144_f16",
   "const_0.7071067811865476_f16",
   "g12_f32_16p0",
   "v_embedding.Unsqueeze_14",
   "embedding.per_layer_model_projection.weight_t",
   "embedding.per_layer_projection_norm.weight",
   "g12_embed/rows_shape"
  ],
  "renamed": {
   "const_1d_1": "g12_embed/const_1d_1"
  },
  "lm_scales_dims": [
   12582912
  ],
  "gather_scales_dims": [
   262144,
   48,
   1
  ],
  "gather_scales_external": {
   "location": "decoder.onnx_data",
   "offset": "1275183104",
   "length": "25165824"
  },
  "table_bits": 4,
  "new_inputs": [
   "input_ids",
   "image_features",
   "ple_rows"
  ],
  "external_kept": [
   "embedding.per_layer_model_projection.weight_t"
  ],
  "moved_to_decoder_data": [
   {
    "name": "embedding.per_layer_model_projection.weight_t",
    "from": "embed.onnx_data",
    "to": "decoder.onnx_data",
    "offset": 1300348928,
    "length": 27525120
   }
  ],
  "lookup_check": {
   "shape": [
    1,
    246,
    1536
   ],
   "bit_identical": true,
   "tokens": 246,
   "per_output": {
    "inputs_embeds": true,
    "per_layer_inputs": true
   }
  }
 }
}