yongqiang commited on
Commit
03f32f8
·
1 Parent(s): 00a9fae

Align gemma4 runtime layout and refresh Python deployment docs

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +1 -0
  2. .gitignore +1 -0
  3. README.md +287 -33
  4. gemma_4_e2b_it_ax650n_axmodel/model.embed_tokens.weight.float32.bin → assets/gemma4_axera_banner.jpg +2 -2
  5. config.json +75 -1
  6. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l0_together.axmodel → gemma4_text_p128_l0_together.axmodel +0 -0
  7. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l10_together.axmodel → gemma4_text_p128_l10_together.axmodel +0 -0
  8. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l11_together.axmodel → gemma4_text_p128_l11_together.axmodel +0 -0
  9. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l12_together.axmodel → gemma4_text_p128_l12_together.axmodel +0 -0
  10. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l13_together.axmodel → gemma4_text_p128_l13_together.axmodel +0 -0
  11. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l14_together.axmodel → gemma4_text_p128_l14_together.axmodel +0 -0
  12. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l15_together.axmodel → gemma4_text_p128_l15_together.axmodel +0 -0
  13. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l16_together.axmodel → gemma4_text_p128_l16_together.axmodel +0 -0
  14. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l17_together.axmodel → gemma4_text_p128_l17_together.axmodel +0 -0
  15. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l18_together.axmodel → gemma4_text_p128_l18_together.axmodel +0 -0
  16. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l19_together.axmodel → gemma4_text_p128_l19_together.axmodel +0 -0
  17. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l1_together.axmodel → gemma4_text_p128_l1_together.axmodel +0 -0
  18. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l20_together.axmodel → gemma4_text_p128_l20_together.axmodel +0 -0
  19. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l21_together.axmodel → gemma4_text_p128_l21_together.axmodel +0 -0
  20. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l22_together.axmodel → gemma4_text_p128_l22_together.axmodel +0 -0
  21. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l23_together.axmodel → gemma4_text_p128_l23_together.axmodel +0 -0
  22. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l24_together.axmodel → gemma4_text_p128_l24_together.axmodel +0 -0
  23. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l25_together.axmodel → gemma4_text_p128_l25_together.axmodel +0 -0
  24. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l26_together.axmodel → gemma4_text_p128_l26_together.axmodel +0 -0
  25. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l27_together.axmodel → gemma4_text_p128_l27_together.axmodel +0 -0
  26. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l28_together.axmodel → gemma4_text_p128_l28_together.axmodel +0 -0
  27. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l29_together.axmodel → gemma4_text_p128_l29_together.axmodel +0 -0
  28. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l2_together.axmodel → gemma4_text_p128_l2_together.axmodel +0 -0
  29. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l30_together.axmodel → gemma4_text_p128_l30_together.axmodel +0 -0
  30. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l31_together.axmodel → gemma4_text_p128_l31_together.axmodel +0 -0
  31. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l32_together.axmodel → gemma4_text_p128_l32_together.axmodel +0 -0
  32. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l33_together.axmodel → gemma4_text_p128_l33_together.axmodel +0 -0
  33. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l34_together.axmodel → gemma4_text_p128_l34_together.axmodel +0 -0
  34. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l3_together.axmodel → gemma4_text_p128_l3_together.axmodel +0 -0
  35. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l4_together.axmodel → gemma4_text_p128_l4_together.axmodel +0 -0
  36. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l5_together.axmodel → gemma4_text_p128_l5_together.axmodel +0 -0
  37. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l6_together.axmodel → gemma4_text_p128_l6_together.axmodel +0 -0
  38. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l7_together.axmodel → gemma4_text_p128_l7_together.axmodel +0 -0
  39. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l8_together.axmodel → gemma4_text_p128_l8_together.axmodel +0 -0
  40. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l9_together.axmodel → gemma4_text_p128_l9_together.axmodel +0 -0
  41. gemma_4_e2b_it_ax650n_axmodel/gemma4_text_post.axmodel → gemma4_text_post.axmodel +0 -0
  42. gemma_4_e2b_it_ax650n_axmodel/model.embed_tokens.weight.npy → gemma4_tokenizer.txt +2 -2
  43. vit_models/gemma4_vision_h336_w480_t70.axmodel → gemma4_vision_h336_w480_t70.axmodel +0 -0
  44. vit_models/gemma4_vision_h480_w672_t140.axmodel → gemma4_vision_h480_w672_t140.axmodel +0 -0
  45. vit_models/gemma4_vision_h672_w960_t280.axmodel → gemma4_vision_h672_w960_t280.axmodel +0 -0
  46. gradio_demo.py +7 -10
  47. infer_axmodel.py +10 -13
  48. gemma_4_e2b_it_ax650n_axmodel/model.embed_tokens.weight.bfloat16.bin → model.embed_tokens.weight.bfloat16.bin +0 -0
  49. gemma_4_e2b_it_ax650n_axmodel/embed_tokens_per_layer.weight.npy → model.embed_tokens_per_layer.weight.npy +0 -0
  50. gemma_4_e2b_it_ax650n_axmodel/per_layer_model_projection.weight.npy → model.per_layer_model_projection.weight.npy +0 -0
.gitattributes CHANGED
@@ -42,3 +42,4 @@ main_axcl_x86 filter=lfs diff=lfs merge=lfs -text
42
  *.jpg filter=lfs diff=lfs merge=lfs -text
43
  *.mp4 filter=lfs diff=lfs merge=lfs -text
44
  tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
42
  *.jpg filter=lfs diff=lfs merge=lfs -text
43
  *.mp4 filter=lfs diff=lfs merge=lfs -text
44
  tokenizer.json filter=lfs diff=lfs merge=lfs -text
45
+ gemma4_tokenizer.txt filter=lfs diff=lfs merge=lfs -text
.gitignore CHANGED
@@ -7,3 +7,4 @@ dist/
7
  .codex
8
  .claude
9
  .gemini
 
 
7
  .codex
8
  .claude
9
  .gemini
10
+ *cache*
README.md CHANGED
@@ -13,7 +13,7 @@ language:
13
  ---
14
 
15
  <p align="center">
16
- <img src="assets/gemma4_axera_banner.png" alt="Gemma4-Axera Banner">
17
  </p>
18
 
19
  # Gemma 4 E2B on AXERA NPU
@@ -26,13 +26,6 @@ Ready-to-run deployment package for `google/gemma-4-E2B-it` on AX650 / NPU3.
26
  - Includes compiled Gemma 4 text `.axmodel` files and Vision `.axmodel` files.
27
  - Supports both text-only chat and single-image multimodal inference.
28
 
29
- ## Conversion References
30
-
31
- If you need the original model files or want to rebuild the deployment artifacts, start with:
32
-
33
- - Original Hugging Face model: [google/gemma-4-E2B-it](https://huggingface.co/google/gemma-4-E2B-it)
34
- - AXERA conversion and deployment workflow: [AXERA-TECH/gemma-4-E2B-it.axera](https://github.com/AXERA-TECH/gemma-4-E2B-it.axera)
35
-
36
  ## Supported Platform
37
 
38
  - [x] AX650 / NPU3
@@ -52,7 +45,7 @@ All measurements below were taken on AX650 / NPU3. `TTFT` stands for time to fir
52
  - `w8a16`: TTFT is approximately `1664 ms`, with a decode throughput of approximately `10.44 tokens/s`.
53
  - `w4a16`: TTFT is approximately `1233.7 ms`, with a decode throughput of approximately `15.22 tokens/s`.
54
 
55
- The packaged text runtime in this release is the `w8a16` build located in `gemma_4_e2b_it_ax650n_axmodel`. The `w4a16` numbers are provided for reference only.
56
 
57
  ## Vision Encoder Latency
58
 
@@ -68,18 +61,275 @@ The packaged text runtime in this release is the `w8a16` build located in `gemma
68
  .
69
  ├── README.md
70
  ├── config.json
 
71
  ├── infer_axmodel.py
72
  ├── gradio_demo.py
73
  ├── assets/
74
  ├── gemma_4_e2b_it_tokenizer/
75
- ├── gemma_4_e2b_it_ax650n_axmodel/
 
 
 
 
 
 
 
 
 
76
  ├── vit_models/
77
  └── utils/
78
  ```
79
 
80
- The runtime scripts auto-detect the packaged directories above. If you keep this layout unchanged, you can run the examples below without passing extra path arguments.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
81
 
82
- ## Runtime Requirements
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
83
 
84
  Install the following packages on the AX board:
85
 
@@ -88,15 +338,18 @@ Install the following packages on the AX board:
88
  - `numpy`
89
  - `ml_dtypes`
90
  - `pillow`
 
91
  - `gradio` for the web demo only
92
 
93
- If your board image ships with an older `transformers` stack, you can use a pure-Python overlay directory instead:
94
 
95
  ```bash
96
  export PYTHONPATH=/path/to/your/gemma4_pydeps:$PYTHONPATH
97
  ```
98
 
99
- ## Quick Start
 
 
100
 
101
  Enter the package directory on the board:
102
 
@@ -130,9 +383,7 @@ answer >> The capital of the United States is **Washington, D.C.**
130
 
131
  ### Multimodal Inference
132
 
133
- Use the sample image shown below: `assets/sample.png`
134
-
135
- ![sample](assets/sample.png)
136
 
137
  Recommended profile: `70` soft tokens at `336x480`.
138
 
@@ -173,13 +424,13 @@ Here's a detailed description:
173
  **Overall Impression:** The image is energetic, bold, and eye-catching, suitable for use as a mascot, icon, or graphic design element.
174
  ```
175
 
176
- The package also includes two additional higher-resolution Vision models:
177
 
178
  | VIT file | Resolution | Soft tokens |
179
  | --- | --- | --- |
180
- | `vit_models/gemma4_vision_h336_w480_t70.axmodel` | `336x480` | `70` |
181
- | `vit_models/gemma4_vision_h480_w672_t140.axmodel` | `480x672` | `140` |
182
- | `vit_models/gemma4_vision_h672_w960_t280.axmodel` | `672x960` | `280` |
183
 
184
  To use a different profile, pass `--vit_model_path` explicitly. The runtime will infer the matching soft-token count from the filename:
185
 
@@ -188,7 +439,7 @@ python3 infer_axmodel.py \
188
  --image_path ./assets/sample.png \
189
  --prompt "Describe this image in detail." \
190
  --system_prompt "" \
191
- --vit_model_path ./vit_models/gemma4_vision_h480_w672_t140.axmodel \
192
  --max_new_tokens 256
193
  ```
194
 
@@ -197,7 +448,7 @@ python3 infer_axmodel.py \
197
  --image_path ./assets/sample.png \
198
  --prompt "Describe this image in detail." \
199
  --system_prompt "" \
200
- --vit_model_path ./vit_models/gemma4_vision_h672_w960_t280.axmodel \
201
  --max_new_tokens 1024
202
  ```
203
 
@@ -243,23 +494,26 @@ python3 gradio_demo.py \
243
 
244
  After the server starts, open `http://<board-ip>:7860` in your browser.
245
 
246
- ## Packaged Runtime Paths
247
 
248
- The release package uses the following default paths:
249
 
250
  - Tokenizer and config: `./gemma_4_e2b_it_tokenizer`
251
- - Text LLM axmodels: `./gemma_4_e2b_it_ax650n_axmodel`
252
- - Vision axmodels: `./vit_models`
253
 
254
  If you move any of these directories, pass the new values with `--hf_model`, `--axmodel_path`, and `--vit_model_path`.
255
 
256
- ## Notes
257
 
258
- - `.axmodel` execution is board-only and is not supported on x86 hosts.
259
- - The default multimodal profile uses `70` image soft tokens and matches the packaged `336x480` Vision model.
260
- - The current text runtime package contains `35` decoder layers and `kv_cache_len=2047`.
261
- - The packaged runtime already includes the embedding and per-layer weight files needed by Gemma 4. Original `model.safetensors` weights are not required for board-side inference.
262
- - Files under `assets/` are demo inputs for inference examples.
 
 
 
263
 
264
  ## Discussion
265
 
 
13
  ---
14
 
15
  <p align="center">
16
+ <img src="assets/gemma4_axera_banner.jpg" alt="Gemma4-Axera Banner">
17
  </p>
18
 
19
  # Gemma 4 E2B on AXERA NPU
 
26
  - Includes compiled Gemma 4 text `.axmodel` files and Vision `.axmodel` files.
27
  - Supports both text-only chat and single-image multimodal inference.
28
 
 
 
 
 
 
 
 
29
  ## Supported Platform
30
 
31
  - [x] AX650 / NPU3
 
45
  - `w8a16`: TTFT is approximately `1664 ms`, with a decode throughput of approximately `10.44 tokens/s`.
46
  - `w4a16`: TTFT is approximately `1233.7 ms`, with a decode throughput of approximately `15.22 tokens/s`.
47
 
48
+ The packaged text runtime in this release is the `w8a16` build. Its text runtime files are packaged at the repository root. The `w4a16` numbers are provided for reference only.
49
 
50
  ## Vision Encoder Latency
51
 
 
61
  .
62
  ├── README.md
63
  ├── config.json
64
+ ├── post_config.json
65
  ├── infer_axmodel.py
66
  ├── gradio_demo.py
67
  ├── assets/
68
  ├── gemma_4_e2b_it_tokenizer/
69
+ ├── gemma4_tokenizer.txt
70
+ ├── gemma4_text_p128_l*.axmodel
71
+ ├── gemma4_text_post.axmodel
72
+ ├── gemma4_vision_h336_w480_t70.axmodel
73
+ ├── gemma4_vision_h480_w672_t140.axmodel
74
+ ├── gemma4_vision_h672_w960_t280.axmodel
75
+ ├── model.embed_tokens_per_layer.weight.npy
76
+ ├── model.embed_tokens.weight.bfloat16.bin
77
+ ├── model.per_layer_model_projection.weight.npy
78
+ ├── model.per_layer_projection_norm.weight.npy
79
  ├── vit_models/
80
  └── utils/
81
  ```
82
 
83
+ This package uses a hybrid layout: the tokenizer stays in a subdirectory, the packaged text runtime files and Vision `.axmodel` files live at the repository root, and `vit_models/` keeps the accompanying Vision metadata JSON files.
84
+
85
+ The Python demo scripts auto-detect the packaged paths above. If you keep this layout unchanged, you can run the Python examples later in this README without passing extra path arguments.
86
+
87
+ ## Sample Image
88
+
89
+ Both the `axllm` flow and the legacy Python demo flow below can use the packaged sample image:
90
+ `assets/sample.png`
91
+
92
+ ![sample](assets/sample.png)
93
+
94
+ ## Direct Inference with `axllm`
95
+
96
+ > The `axllm` workflow is still being refined. The instructions below reflect the current validated flow and may be adjusted as the packaging continues to evolve.
97
+
98
+ ### Download the Model Package
99
+
100
+ Download the release package from Hugging Face:
101
+
102
+ ```shell
103
+ mkdir -p AXERA-TECH/gemma-4-E2B-it
104
+ cd AXERA-TECH/gemma-4-E2B-it
105
+ hf download AXERA-TECH/gemma-4-E2B-it --local-dir .
106
+ ```
107
+
108
+ ### Install `axllm`
109
+
110
+ Option 1: clone the repository and run the installer:
111
+
112
+ ```shell
113
+ git clone -b axllm https://github.com/AXERA-TECH/ax-llm.git
114
+ cd ax-llm
115
+ ./install.sh
116
+ ```
117
+
118
+ Option 2: install with a one-line command (default branch: `axllm`):
119
+
120
+ ```shell
121
+ curl -fsSL https://raw.githubusercontent.com/AXERA-TECH/ax-llm/axllm/install.sh | bash
122
+ ```
123
+
124
+ Option 3: download the prebuilt binary from GitHub Actions CI:
125
+
126
+ If you do not have a local build environment, download the latest CI-generated `axllm` binary from GitHub Actions:
127
+ `https://github.com/AXERA-TECH/ax-llm/actions?query=branch%3Aaxllm`
128
+ Then run:
129
+
130
+ ```shell
131
+ chmod +x axllm
132
+ sudo mv axllm /usr/bin/axllm
133
+ ```
134
 
135
+ ### Run on the Board
136
+
137
+ The package root is already arranged for `axllm`, so no extra runtime path arguments are required.
138
+
139
+ Note: the command below assumes you run it from the parent directory of `AXERA-TECH/gemma-4-E2B-it`. If you are already inside the package directory, use `axllm run .` instead.
140
+
141
+ For multimodal testing, you can use the sample image shown above: `./assets/sample.png`.
142
+
143
+ ```bash
144
+ $ axllm run AXERA-TECH/gemma-4-E2B-it
145
+
146
+ # output log example:
147
+ 15:04:24.522 INF Init:890 | LLM init start
148
+ 15:04:24.522 INF Init:905 | shared kv enabled: num_kv_shared_layers=20
149
+ tokenizer_type = 3
150
+ huggingface tokenizer mode = space_replace_bpe
151
+ 31% | ########## | 12 / 38 [4.47s<14.16s, 2.68 count/s] init 10 axmodel ok,remain_cmm(6047 MB 34% | ########## | 13 / 38 [4.61s<13.48s, 2.82 count/s] init 11 axmodel ok,remain_cmm(5992 MB 36% | ########### | 14 / 38 [4.78s<12.98s, 2.93 count/s] init 12 axmodel ok,remain_cmm(5937 MB 39% | ############ | 15 / 38 [4.93s<12.49s, 3.04 count/s] init 13 axmodel ok,remain_cmm(5882 MB 42% | ############# | 16 / 38 [5.09s<12.09s, 3.14 count/s] init 14 axmodel ok,remain_cmm(5813 MB 44% | ############## | 17 / 38 [5.28s<11.80s, 3.22 count/s] init 15 axmodel ok,remain_cmm(5727 MB 47% | ############### | 18 / 38 [5.50s<11.61s, 3.27 count/s] init 16 axmodel ok,remain_cmm(5642 MB 50% | ################ | 19 / 38 [5.69s<11.38s, 3.34 count/s] init 17 axmodel ok,remain_cmm(5557 MB 52% | ################ | 20 / 38 [5.91s<11.22s, 3.39 count/s] init 18 axmodel ok,remain_cmm(5471 MB 55% | ################# | 21 / 38 [6.11s<11.06s, 3.44 count/s] init 19 axmodel ok,remain_cmm(5373 MB 57% | ################## | 22 / 38 [6.31s<10.89s, 3.49 count/s] init 20 axmodel ok,remain_cmm(5287 MB 60% | ################### | 23 / 38 [6.53s<10.79s, 3.52 count/s] init 21 axmodel ok,remain_cmm(5202 MB 63% | #################### | 24 / 38 [6.75s<10.69s, 3.56 count/s] init 22 axmodel ok,remain_cmm(5117 MB 65% | ##################### | 25 / 38 [6.96s<10.58s, 3.59 count/s] init 23 axmodel ok,remain_cmm(5031 MB 68% | ##################### | 26 / 38 [7.18s<10.50s, 3.62 count/s] init 24 axmodel ok,remain_cmm(4933 MB 71% | ###################### | 27 / 38 [7.40s<10.41s, 3.65 count/s] init 25 axmodel ok,remain_cmm(4847 MB 73% | ####################### | 28 / 38 [7.62s<10.34s, 3.67 count/s] init 26 axmodel ok,remain_cmm(4762 MB 76% | ######################## | 29 / 38 [7.85s<10.28s, 3.70 count/s] init 27 axmodel ok,remain_cmm(4676 MB 78% | ######################### | 30 / 38 [8.13s<10.29s, 3.69 count/s] init 28 axmodel ok,remain_cmm(4591 MB 81% | ########################## | 31 / 38 [8.36s<10.25s, 3.71 count/s] init 29 axmodel ok,remain_cmm(4492 MB 84% | ########################## | 32 / 38 [8.60s<10.21s, 3.72 count/s] init 30 axmodel ok,remain_cmm(4407 MB 86% | ########################### | 33 / 38 [8.86s<10.21s, 3.72 count/s] init 31 axmodel ok,remain_cmm(4322 MB 89% | ############################ | 34 / 38 [9.11s<10.18s, 3.73 count/s] init 32 axmodel ok,remain_cmm(4236 MB 92% | ############################# | 35 / 38 [9.36s<10.16s, 3.74 count/s] init 33 axmodel ok,remain_cmm(4151 MB 94% | ############################## | 36 / 38 [9.62s<10.16s, 3.74 count/s] init 34 axmodel ok,remain_cmm(4052 MB 97% | ############################### | 37 / 38 [10.03s<10.30s, 3.69 count/s] init post axmodel ok,remain_cmm(3632 MB)
152
+ 15:04:34.551 INF Init:1045 | max_token_len : 2047
153
+ 15:04:34.551 INF Init:1048 | kv_cache_size : 256, kv_cache_num: 2047
154
+ 15:04:34.551 INF init_groups_from_model:606 | prefill_token_num : 128
155
+ 15:04:34.551 INF init_groups_from_model:820 | decode grp: 0, gid: 0, max_token_len : 2047
156
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 0, gid: 1, history_cap: 0, total_cap: 128, symbolic_cap: 1
157
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 1, gid: 2, history_cap: 128, total_cap: 256, symbolic_cap: 128
158
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 2, gid: 3, history_cap: 256, total_cap: 384, symbolic_cap: 256
159
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 3, gid: 4, history_cap: 384, total_cap: 512, symbolic_cap: 384
160
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 4, gid: 5, history_cap: 512, total_cap: 640, symbolic_cap: 512
161
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 5, gid: 6, history_cap: 640, total_cap: 768, symbolic_cap: 640
162
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 6, gid: 7, history_cap: 768, total_cap: 896, symbolic_cap: 768
163
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 7, gid: 8, history_cap: 896, total_cap: 1024, symbolic_cap: 896
164
+ 15:04:34.551 INF init_groups_from_model:824 | prefill grp: 8, gid: 9, history_cap: 1024, total_cap: 1152, symbolic_cap: 1024
165
+ 15:04:34.551 INF init_groups_from_model:831 | prefill_max_token_num : 1152
166
+ 15:04:34.551 INF Init:27 | LLaMaEmbedSelector use mmap
167
+ 100% | ################################ | 38 / 38 [10.03s<10.03s, 3.79 count/s] embed_selector init ok
168
+ 15:04:34.567 INF Init:475 | Gemma4 per-layer helper enabled: vocab=262144 hidden=1536 layers=35 per_layer=256 pad=0
169
+ 15:04:34.727 INF Init:785 | Gemma4-VL token ids: image_pad=258880 video_pad=258884
170
+ 15:04:34.727 INF Init:792 | VisionModule init ok: type=Gemma4VL, tokens_per_block=70, embed_size=1536, out_dtype=fp32
171
+ 15:04:34.727 WRN Init:801 | Vision preprocess backend: SimpleCV (OpenCV not found at build time; minor differences vs OpenCV are possible)
172
+ 15:04:34.729 INF load_config:282 | load config:
173
+ 15:04:34.729 INF load_config:282 | {
174
+ 15:04:34.729 INF load_config:282 | "enable_repetition_penalty": false,
175
+ 15:04:34.729 INF load_config:282 | "enable_temperature": true,
176
+ 15:04:34.729 INF load_config:282 | "enable_top_k_sampling": false,
177
+ 15:04:34.729 INF load_config:282 | "enable_top_p_sampling": true,
178
+ 15:04:34.729 INF load_config:282 | "penalty_window": 64,
179
+ 15:04:34.729 INF load_config:282 | "repetition_penalty": 1.0,
180
+ 15:04:34.729 INF load_config:282 | "temperature": 1.0,
181
+ 15:04:34.729 INF load_config:282 | "top_k": 64,
182
+ 15:04:34.729 INF load_config:282 | "top_p": 0.95
183
+ 15:04:34.729 INF load_config:282 | }
184
+ 15:04:34.729 INF Init:1139 | LLM init ok
185
+ Commands:
186
+ /q, /exit 退出
187
+ /reset 重置 kvcache
188
+ /dd 删除一轮对话
189
+ /pp 打印历史对话
190
+ Ctrl+C: 停止当前生成
191
+ VLM enabled: after each prompt, input media path (empty = text-only). Use "video:<frames_dir>" for video, "audio:<file>" for reserved audio placeholder.
192
+ ----------------------------------------
193
+ prompt >> who are you?
194
+ media >>
195
+ 15:04:39.368 INF SetKVCache:1437 | decode_grpid:0 prefill_grpid:1 history_cap:0 total_cap:128 symbolic_cap:1 precompute_len:0 input_num_token:24 prefer_symbolic_group:0
196
+ 15:04:39.368 INF SetKVCache:1458 | current prefill_max_token_num:1152
197
+ 15:04:39.408 INF SetKVCache:1462 | first run
198
+ 15:04:39.409 INF Run:1553 | input token num : 24, prefill_split_num : 1
199
+ 15:04:39.482 INF Run:1640 | prefill chunk p=0 history_len=0 grpid=1 kv_cache_num=0 input_tokens=24
200
+ 15:04:39.483 INF Run:1665 | prefill indices shape: p=0 idx_elems=128 idx_rows=1 pos_rows=0
201
+ 15:04:39.764 INF Run:1837 | ttft: 355.37 ms
202
+ I am Gemma 4, a Large Language Model developed by Google DeepMind. I am an open weights model.
203
+
204
+ 15:04:44.087 NTC Run:2103 | hit eos,decode avg 5.09 token/s
205
+ 15:04:44.088 INF GetKVCache:1408 | precompute_len:47, remaining:1105
206
+ prompt >> Please describe the image in detail.
207
+ media >> /root/yongqiang/auto_model_deployment/gemma-4-E2B-it/assets/sample.png
208
+ 15:06:14.416 INF EncodeForContent:1122 | vision cache hit (disk): /root/yongqiang/auto_model_deployment/gemma-4-E2B-it/assets/sample.png
209
+ 15:06:14.416 INF EncodeForContent:1131 | vision cache hit (mem): /root/yongqiang/auto_model_deployment/gemma-4-E2B-it/assets/sample.png
210
+ 15:06:14.419 INF SetKVCache:1437 | decode_grpid:0 prefill_grpid:3 history_cap:256 total_cap:384 symbolic_cap:256 precompute_len:47 input_num_token:94 prefer_symbolic_group:1
211
+ 15:06:14.419 INF SetKVCache:1458 | current prefill_max_token_num:1024
212
+ 15:06:14.429 INF Run:1553 | input token num : 94, prefill_split_num : 1
213
+ 15:06:14.703 INF Run:1640 | prefill chunk p=0 history_len=47 grpid=3 kv_cache_num=256 input_tokens=94
214
+ 15:06:14.703 INF Run:1665 | prefill indices shape: p=0 idx_elems=128 idx_rows=1 pos_rows=0
215
+ 15:06:15.027 INF Run:1837 | ttft: 597.95 ms
216
+ I see an image of a cartoon character that resembles a cooked or stylized lobster.
217
+
218
+ Here is a detailed description of the image:
219
+
220
+ * **Subject:** The central subject is a bright red, stylized lobster.
221
+ * **Style:** The illustration is highly cartoonish and vibrant, featuring thick outlines and bright, saturated colors, suggesting a playful or energetic style.
222
+ * **Features:**
223
+ * The lobster has large, expressive eyes and a wide, toothy grin, giving it a mischievous or energetic expression.
224
+ * Its claws (pincers) are prominent and stylized.
225
+ * The body is segmented, typical of a lobster, but rendered in a simplified, bold manner.
226
+ * It has a curved, slightly exaggerated posture.
227
+ * **Outline/Background:** The character is set against a plain white background. The image has a glossy or sticker-like finish, indicated by a slight shadow effect or outline around the character, suggesting it might be a graphic or sticker design.
228
+ * **Overall Impression:** The image is energetic, bold, and fun, clearly designed as a mascot or a character illustration.
229
+
230
+ 15:07:09.593 NTC Run:2103 | hit eos,decode avg 4.33 token/s
231
+ 15:07:09.593 INF GetKVCache:1408 | precompute_len:378, remaining:774
232
+ ```
233
+
234
+
235
+ ### Serve with `axllm`
236
+
237
+ To launch the packaged model through the local `axllm` service:
238
+
239
+ Note: the command below assumes you run it from the parent directory of `AXERA-TECH/gemma-4-E2B-it`. If you are already inside the package directory, use `axllm serve . --port 8000` instead.
240
+
241
+ ```bash
242
+ $ axllm serve AXERA-TECH/gemma-4-E2B-it --port 8000
243
+ # output log example:
244
+ 16:22:21.336 INF Init:890 | LLM init start
245
+ 16:22:21.336 INF Init:905 | shared kv enabled: num_kv_shared_layers=20
246
+ tokenizer_type = 3
247
+ huggingface tokenizer mode = space_replace_bpe
248
+ 13% | #### | 5 / 38 [10.08s<76.64s, 0.50 count/s] init 3 axmodel ok,remain_cmm(4704 MB 15% | ##### | 6 / 38 [12.32s<78.01s, 0.49 count/s] init 4 axmodel ok,remain_cmm(4635 MB 18% | ##### | 7 / 38 [17.98s<97.59s, 0.39 count/s] init 5 axmodel ok,remain_cmm(4580 MB 21% | ###### | 8 / 38 [18.54s<88.05s, 0.43 count/s] init 6 axmodel ok,remain_cmm(4525 MB 23% | ####### | 9 / 38 [19.06s<80.49s, 0.47 count/s] init 7 axmodel ok,remain_cmm(4470 MB 26% | ######## | 10 / 38 [19.66s<74.70s, 0.51 count/s] init 8 axmodel ok,remain_cmm(4415 MB 28% | ######### | 11 / 38 [20.41s<70.49s, 0.54 count/s] init 9 axmodel ok,remain_cmm(4346 MB 31% | ########## | 12 / 38 [20.81s<65.91s, 0.58 count/s] init 10 axmodel ok,remain_cmm(4291 M 34% | ########## | 13 / 38 [21.22s<62.04s, 0.61 count/s] init 11 axmodel ok,remain_cmm(4236 M 36% | ########### | 14 / 38 [21.89s<59.42s, 0.64 count/s] init 12 axmodel ok,remain_cmm(4182 M 39% | ############ | 15 / 38 [22.24s<56.34s, 0.67 count/s] init 13 axmodel ok,remain_cmm(4127 M 42% | ############# | 16 / 38 [22.56s<53.57s, 0.71 count/s] init 14 axmodel ok,remain_cmm(4057 M 44% | ############## | 17 / 38 [23.11s<51.66s, 0.74 count/s] init 15 axmodel ok,remain_cmm(3972 M 47% | ############### | 18 / 38 [23.54s<49.69s, 0.76 count/s] init 16 axmodel ok,remain_cmm(3887 M 50% | ################ | 19 / 38 [24.28s<48.56s, 0.78 count/s] init 17 axmodel ok,remain_cmm(3801 M 52% | ################ | 20 / 38 [24.63s<46.80s, 0.81 count/s] init 18 axmodel ok,remain_cmm(3716 M 55% | ################# | 21 / 38 [24.92s<45.08s, 0.84 count/s] init 19 axmodel ok,remain_cmm(3617 M 57% | ################## | 22 / 38 [25.67s<44.33s, 0.86 count/s] init 20 axmodel ok,remain_cmm(3532 M 60% | ################### | 23 / 38 [26.33s<43.50s, 0.87 count/s] init 21 axmodel ok,remain_cmm(3447 M 63% | #################### | 24 / 38 [27.17s<43.02s, 0.88 count/s] init 22 axmodel ok,remain_cmm(3361 M 65% | ##################### | 25 / 38 [28.33s<43.06s, 0.88 count/s] init 23 axmodel ok,remain_cmm(3276 M 68% | ##################### | 26 / 38 [29.70s<43.41s, 0.88 count/s] init 24 axmodel ok,remain_cmm(3177 M 71% | ###################### | 27 / 38 [30.89s<43.48s, 0.87 count/s] init 25 axmodel ok,remain_cmm(3092 M 73% | ####################### | 28 / 38 [32.16s<43.65s, 0.87 count/s] init 26 axmodel ok,remain_cmm(3006 M 76% | ######################## | 29 / 38 [33.32s<43.67s, 0.87 count/s] init 27 axmodel ok,remain_cmm(2921 M 78% | ######################### | 30 / 38 [34.43s<43.61s, 0.87 count/s] init 28 axmodel ok,remain_cmm(2836 M 81% | ########################## | 31 / 38 [35.69s<43.75s, 0.87 count/s] init 29 axmodel ok,remain_cmm(2737 M 84% | ########################## | 32 / 38 [36.84s<43.75s, 0.87 count/s] init 30 axmodel ok,remain_cmm(2652 M 86% | ########################### | 33 / 38 [37.75s<43.47s, 0.87 count/s] init 31 axmodel ok,remain_cmm(2566 M 89% | ############################ | 34 / 38 [38.44s<42.96s, 0.88 count/s] init 32 axmodel ok,remain_cmm(2481 M 92% | ############################# | 35 / 38 [39.06s<42.41s, 0.90 count/s] init 33 axmodel ok,remain_cmm(2396 M 94% | ############################## | 36 / 38 [39.44s<41.63s, 0.91 count/s] init 34 axmodel ok,remain_cmm(2297 M 97% | ############################### | 37 / 38 [41.12s<42.23s, 0.90 count/s] init post axmodel ok,remain_cmm(1877 MB)
249
+ 16:23:02.455 INF Init:1045 | max_token_len : 2047
250
+ 16:23:02.455 INF Init:1048 | kv_cache_size : 256, kv_cache_num: 2047
251
+ 16:23:02.455 INF init_groups_from_model:606 | prefill_token_num : 128
252
+ 16:23:02.455 INF init_groups_from_model:820 | decode grp: 0, gid: 0, max_token_len : 2047
253
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 0, gid: 1, history_cap: 0, total_cap: 128, symbolic_cap: 1
254
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 1, gid: 2, history_cap: 128, total_cap: 256, symbolic_cap: 128
255
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 2, gid: 3, history_cap: 256, total_cap: 384, symbolic_cap: 256
256
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 3, gid: 4, history_cap: 384, total_cap: 512, symbolic_cap: 384
257
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 4, gid: 5, history_cap: 512, total_cap: 640, symbolic_cap: 512
258
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 5, gid: 6, history_cap: 640, total_cap: 768, symbolic_cap: 640
259
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 6, gid: 7, history_cap: 768, total_cap: 896, symbolic_cap: 768
260
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 7, gid: 8, history_cap: 896, total_cap: 1024, symbolic_cap: 896
261
+ 16:23:02.455 INF init_groups_from_model:824 | prefill grp: 8, gid: 9, history_cap: 1024, total_cap: 1152, symbolic_cap: 1024
262
+ 16:23:02.455 INF init_groups_from_model:831 | prefill_max_token_num : 1152
263
+ 16:23:02.455 INF Init:27 | LLaMaEmbedSelector use mmap
264
+ 100% | ################################ | 38 / 38 [41.12s<41.12s, 0.92 count/s] embed_selector init ok
265
+ 16:23:02.472 INF Init:475 | Gemma4 per-layer helper enabled: vocab=262144 hidden=1536 layers=35 per_layer=256 pad=0
266
+ 16:23:03.400 INF Init:785 | Gemma4-VL token ids: image_pad=258880 video_pad=258884
267
+ 16:23:03.400 INF Init:792 | VisionModule init ok: type=Gemma4VL, tokens_per_block=70, embed_size=1536, out_dtype=fp32
268
+ 16:23:03.400 WRN Init:801 | Vision preprocess backend: SimpleCV (OpenCV not found at build time; minor differences vs OpenCV are possible)
269
+ 16:23:03.404 INF load_config:282 | load config:
270
+ 16:23:03.404 INF load_config:282 | {
271
+ 16:23:03.404 INF load_config:282 | "enable_repetition_penalty": false,
272
+ 16:23:03.404 INF load_config:282 | "enable_temperature": true,
273
+ 16:23:03.404 INF load_config:282 | "enable_top_k_sampling": false,
274
+ 16:23:03.404 INF load_config:282 | "enable_top_p_sampling": true,
275
+ 16:23:03.404 INF load_config:282 | "penalty_window": 64,
276
+ 16:23:03.404 INF load_config:282 | "repetition_penalty": 1.0,
277
+ 16:23:03.404 INF load_config:282 | "temperature": 1.0,
278
+ 16:23:03.404 INF load_config:282 | "top_k": 64,
279
+ 16:23:03.404 INF load_config:282 | "top_p": 0.95
280
+ 16:23:03.404 INF load_config:282 | }
281
+ 16:23:03.404 INF Init:1139 | LLM init ok
282
+ Starting server on port 8000 with model 'AXERA-TECH/gemma-4-E2B-it'...
283
+ API URLs:
284
+ GET http://127.0.0.1:8000/health
285
+ GET http://127.0.0.1:8000/v1/models
286
+ POST http://127.0.0.1:8000/v1/chat/completions
287
+ GET http://10.168.232.217:8000/health
288
+ GET http://10.168.232.217:8000/v1/models
289
+ POST http://10.168.232.217:8000/v1/chat/completions
290
+ GET http://172.17.0.1:8000/health
291
+ GET http://172.17.0.1:8000/v1/models
292
+ POST http://172.17.0.1:8000/v1/chat/completions
293
+ Aliases:
294
+ GET http://127.0.0.1:8000/models
295
+ POST http://127.0.0.1:8000/chat/completions
296
+ GET http://10.168.232.217:8000/models
297
+ POST http://10.168.232.217:8000/chat/completions
298
+ GET http://172.17.0.1:8000/models
299
+ POST http://172.17.0.1:8000/chat/completions
300
+ OpenAI API Server starting on http://0.0.0.0:8000
301
+ Max concurrency: 1
302
+ Models: AXERA-TECH/gemma-4-E2B-it
303
+ ```
304
+
305
+ You can then send requests to the server using the API endpoints shown in the log. For example, to check the health status and list the available models:
306
+
307
+ ```sh
308
+ $ curl http://127.0.0.1:8000/health
309
+ $ curl http://127.0.0.1:8000/v1/models
310
+
311
+ # Example output:
312
+ root@ax650 ~ # curl http://127.0.0.1:8000/health
313
+ {
314
+ "concurrency": 0,
315
+ "max_concurrency": 1,
316
+ "status": "healthy"
317
+ }
318
+ root@ax650 ~ # curl http://127.0.0.1:8000/v1/models
319
+ {
320
+ "data": [
321
+ {
322
+ "created": 1777019000,
323
+ "id": "AXERA-TECH/gemma-4-E2B-it",
324
+ "object": "model",
325
+ "owned_by": "openai-api"
326
+ }
327
+ ],
328
+ "object": "list"
329
+ }
330
+ ```
331
+
332
+ ## Python Runtime Requirements
333
 
334
  Install the following packages on the AX board:
335
 
 
338
  - `numpy`
339
  - `ml_dtypes`
340
  - `pillow`
341
+ - `torch`
342
  - `gradio` for the web demo only
343
 
344
+ Before running any Python demo command in this package, make sure the Python dependency overlay is visible in `PYTHONPATH`:
345
 
346
  ```bash
347
  export PYTHONPATH=/path/to/your/gemma4_pydeps:$PYTHONPATH
348
  ```
349
 
350
+ If your board image ships with an older `transformers` stack, this pure-Python overlay is the recommended way to supply the required runtime dependencies.
351
+
352
+ ## Legacy Python Demo Flow
353
 
354
  Enter the package directory on the board:
355
 
 
383
 
384
  ### Multimodal Inference
385
 
386
+ Use the sample image shown above: `assets/sample.png`
 
 
387
 
388
  Recommended profile: `70` soft tokens at `336x480`.
389
 
 
424
  **Overall Impression:** The image is energetic, bold, and eye-catching, suitable for use as a mascot, icon, or graphic design element.
425
  ```
426
 
427
+ In addition to the default `t70` profile, the package also includes two higher-resolution Vision models:
428
 
429
  | VIT file | Resolution | Soft tokens |
430
  | --- | --- | --- |
431
+ | `gemma4_vision_h336_w480_t70.axmodel` | `336x480` | `70` |
432
+ | `gemma4_vision_h480_w672_t140.axmodel` | `480x672` | `140` |
433
+ | `gemma4_vision_h672_w960_t280.axmodel` | `672x960` | `280` |
434
 
435
  To use a different profile, pass `--vit_model_path` explicitly. The runtime will infer the matching soft-token count from the filename:
436
 
 
439
  --image_path ./assets/sample.png \
440
  --prompt "Describe this image in detail." \
441
  --system_prompt "" \
442
+ --vit_model_path ./gemma4_vision_h480_w672_t140.axmodel \
443
  --max_new_tokens 256
444
  ```
445
 
 
448
  --image_path ./assets/sample.png \
449
  --prompt "Describe this image in detail." \
450
  --system_prompt "" \
451
+ --vit_model_path ./gemma4_vision_h672_w960_t280.axmodel \
452
  --max_new_tokens 1024
453
  ```
454
 
 
494
 
495
  After the server starts, open `http://<board-ip>:7860` in your browser.
496
 
497
+ ## Packaged Python Runtime Paths
498
 
499
+ The Python demo scripts use the following default paths:
500
 
501
  - Tokenizer and config: `./gemma_4_e2b_it_tokenizer`
502
+ - Text LLM runtime root: `./`
503
+ - Vision axmodels: `./`
504
 
505
  If you move any of these directories, pass the new values with `--hf_model`, `--axmodel_path`, and `--vit_model_path`.
506
 
507
+ For the Python demo flow, `--axmodel_path` should point to the directory that contains the text runtime files such as `gemma4_text_p128_l*.axmodel`, `gemma4_text_post.axmodel`, `model.embed_tokens.weight.bfloat16.bin`, and the `model.*per_layer*.npy` files.
508
 
509
+ These path arguments apply to the Python demo flow only. The `axllm` flow reads the same root-level runtime files packaged in this repository.
510
+
511
+ ## Conversion References
512
+
513
+ If you need the original model files or want to rebuild the deployment artifacts, start with:
514
+
515
+ - Original Hugging Face model: [google/gemma-4-E2B-it](https://huggingface.co/google/gemma-4-E2B-it)
516
+ - AXERA conversion and deployment workflow: [AXERA-TECH/gemma-4-E2B-it.axera](https://github.com/AXERA-TECH/gemma-4-E2B-it.axera)
517
 
518
  ## Discussion
519
 
gemma_4_e2b_it_ax650n_axmodel/model.embed_tokens.weight.float32.bin → assets/gemma4_axera_banner.jpg RENAMED
File without changes
config.json CHANGED
@@ -1 +1,75 @@
1
- {}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "system_prompt": "You are a helpful assistant.",
3
+ "model_name": "AXERA-TECH/gemma-4-E2B-it",
4
+ "url_tokenizer_model": "gemma4_tokenizer.txt",
5
+ "tokenizer_type": "Gemma4VL",
6
+ "post_config_path": "post_config.json",
7
+ "template_filename_axmodel": "gemma4_text_p128_l%d_together.axmodel",
8
+ "axmodel_num": 35,
9
+ "filename_post_axmodel": "gemma4_text_post.axmodel",
10
+ "filename_tokens_embed": "model.embed_tokens.weight.bfloat16.bin",
11
+ "tokens_embed_num": 262144,
12
+ "tokens_embed_size": 1536,
13
+ "text_config": {
14
+ "hidden_size": 1536,
15
+ "num_hidden_layers": 35,
16
+ "num_key_value_heads": 1,
17
+ "head_dim": 256,
18
+ "global_head_dim": 512,
19
+ "num_kv_shared_layers": 20,
20
+ "layer_types": [
21
+ "sliding_attention",
22
+ "sliding_attention",
23
+ "sliding_attention",
24
+ "sliding_attention",
25
+ "full_attention",
26
+ "sliding_attention",
27
+ "sliding_attention",
28
+ "sliding_attention",
29
+ "sliding_attention",
30
+ "full_attention",
31
+ "sliding_attention",
32
+ "sliding_attention",
33
+ "sliding_attention",
34
+ "sliding_attention",
35
+ "full_attention",
36
+ "sliding_attention",
37
+ "sliding_attention",
38
+ "sliding_attention",
39
+ "sliding_attention",
40
+ "full_attention",
41
+ "sliding_attention",
42
+ "sliding_attention",
43
+ "sliding_attention",
44
+ "sliding_attention",
45
+ "full_attention",
46
+ "sliding_attention",
47
+ "sliding_attention",
48
+ "sliding_attention",
49
+ "sliding_attention",
50
+ "full_attention",
51
+ "sliding_attention",
52
+ "sliding_attention",
53
+ "sliding_attention",
54
+ "sliding_attention",
55
+ "full_attention"
56
+ ]
57
+ },
58
+ "pad_token_id": 0,
59
+ "hidden_size_per_layer_input": 256,
60
+ "rms_norm_eps": 1e-06,
61
+ "filename_tokens_embed_per_layer": "model.embed_tokens_per_layer.weight.npy",
62
+ "filename_per_layer_model_projection": "model.per_layer_model_projection.weight.npy",
63
+ "filename_per_layer_projection_norm": "model.per_layer_projection_norm.weight.npy",
64
+ "use_mmap_load_embed": true,
65
+ "vlm_type": "Gemma4VL",
66
+ "filename_image_encoder_axmodel": "gemma4_vision_h336_w480_t70.axmodel",
67
+ "vision_width": 480,
68
+ "vision_height": 336,
69
+ "vision_patch_size": 16,
70
+ "vision_cache_dir": "vision_cache",
71
+ "use_mmap_load_layer": true,
72
+ "devices": [
73
+ 0
74
+ ]
75
+ }
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l0_together.axmodel → gemma4_text_p128_l0_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l10_together.axmodel → gemma4_text_p128_l10_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l11_together.axmodel → gemma4_text_p128_l11_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l12_together.axmodel → gemma4_text_p128_l12_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l13_together.axmodel → gemma4_text_p128_l13_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l14_together.axmodel → gemma4_text_p128_l14_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l15_together.axmodel → gemma4_text_p128_l15_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l16_together.axmodel → gemma4_text_p128_l16_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l17_together.axmodel → gemma4_text_p128_l17_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l18_together.axmodel → gemma4_text_p128_l18_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l19_together.axmodel → gemma4_text_p128_l19_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l1_together.axmodel → gemma4_text_p128_l1_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l20_together.axmodel → gemma4_text_p128_l20_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l21_together.axmodel → gemma4_text_p128_l21_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l22_together.axmodel → gemma4_text_p128_l22_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l23_together.axmodel → gemma4_text_p128_l23_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l24_together.axmodel → gemma4_text_p128_l24_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l25_together.axmodel → gemma4_text_p128_l25_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l26_together.axmodel → gemma4_text_p128_l26_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l27_together.axmodel → gemma4_text_p128_l27_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l28_together.axmodel → gemma4_text_p128_l28_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l29_together.axmodel → gemma4_text_p128_l29_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l2_together.axmodel → gemma4_text_p128_l2_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l30_together.axmodel → gemma4_text_p128_l30_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l31_together.axmodel → gemma4_text_p128_l31_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l32_together.axmodel → gemma4_text_p128_l32_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l33_together.axmodel → gemma4_text_p128_l33_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l34_together.axmodel → gemma4_text_p128_l34_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l3_together.axmodel → gemma4_text_p128_l3_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l4_together.axmodel → gemma4_text_p128_l4_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l5_together.axmodel → gemma4_text_p128_l5_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l6_together.axmodel → gemma4_text_p128_l6_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l7_together.axmodel → gemma4_text_p128_l7_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l8_together.axmodel → gemma4_text_p128_l8_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_p128_l9_together.axmodel → gemma4_text_p128_l9_together.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/gemma4_text_post.axmodel → gemma4_text_post.axmodel RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/model.embed_tokens.weight.npy → gemma4_tokenizer.txt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:39202a6627a585f06b0852d267311d3a063fb8d610c1c106ec403d05d42afba0
3
- size 1610612864
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:90603c2c15f0d202d63c5c7767e4787e7e3909d74ee2c47f93914d3575dbd0ef
3
+ size 17165772
vit_models/gemma4_vision_h336_w480_t70.axmodel → gemma4_vision_h336_w480_t70.axmodel RENAMED
File without changes
vit_models/gemma4_vision_h480_w672_t140.axmodel → gemma4_vision_h480_w672_t140.axmodel RENAMED
File without changes
vit_models/gemma4_vision_h672_w960_t280.axmodel → gemma4_vision_h672_w960_t280.axmodel RENAMED
File without changes
gradio_demo.py CHANGED
@@ -23,6 +23,8 @@ from utils.gemma4_multimodal import resolve_resize
23
  from utils.gemma4_multimodal import resize_image
24
  from utils.gemma4_multimodal import to_numpy_fp32
25
  from utils.infer_func import InferManager
 
 
26
  from utils.vision_output import describe_output_shapes
27
  from utils.vision_output import select_vit_output
28
 
@@ -53,14 +55,7 @@ def _default_hf_model() -> str:
53
 
54
  def _default_axmodel_path() -> str:
55
  script_dir = Path(__file__).resolve().parent
56
- for candidate in [
57
- script_dir / "gemma_4_e2b_it_ax650n_axmodel",
58
- script_dir / "gemma_4_e2b_it_ax650n_w4a16_axmodel",
59
- script_dir / "gemma-4-E2B-it_axmodel",
60
- ]:
61
- if candidate.exists():
62
- return str(candidate)
63
- return str(script_dir / "gemma_4_e2b_it_ax650n_axmodel")
64
 
65
 
66
  def _default_vit_model_path() -> str:
@@ -68,7 +63,9 @@ def _default_vit_model_path() -> str:
68
  resize_h, resize_w, expected_tokens = resolve_resize(DEFAULT_MAX_SOFT_TOKENS)
69
  stem = f"gemma4_vision_h{resize_h}_w{resize_w}_t{expected_tokens}"
70
  candidates = [
 
71
  script_dir / "vit_models" / f"{stem}.axmodel",
 
72
  script_dir / "vit_models" / f"{stem}.onnx",
73
  script_dir.parent / "model_convert" / "compiled_output" / f"{stem}.axmodel",
74
  script_dir.parent / "model_convert" / "vit-models" / f"{stem}.onnx",
@@ -108,7 +105,7 @@ class Gemma4GradioDemo:
108
  self.processor = load_processor(hf_model)
109
  self.tokenizer = self.processor.tokenizer
110
  self.config = load_text_runtime_config(hf_model)
111
- self.embeds = np.load(os.path.join(axmodel_path, "model.embed_tokens.weight.npy"))
112
  self.axmodel_path = axmodel_path
113
  self.vit_model_path = vit_model_path
114
  self.max_seq_len = max_seq_len
@@ -345,7 +342,7 @@ def main():
345
  parser.add_argument("--hf_model", type=str, default=_default_hf_model(),
346
  help="Path to Gemma 4 tokenizer/config directory")
347
  parser.add_argument("--axmodel_path", type=str, default=_default_axmodel_path(),
348
- help="Path to compiled LLM axmodel folder")
349
  parser.add_argument("--vit_model_path", type=str, default=_default_vit_model_path(),
350
  help="Path to Gemma 4 vision ONNX model or .axmodel")
351
  parser.add_argument("--port", type=int, default=7860, help="Gradio server port")
 
23
  from utils.gemma4_multimodal import resize_image
24
  from utils.gemma4_multimodal import to_numpy_fp32
25
  from utils.infer_func import InferManager
26
+ from utils.runtime_layout import default_axmodel_path
27
+ from utils.runtime_layout import load_text_embeddings
28
  from utils.vision_output import describe_output_shapes
29
  from utils.vision_output import select_vit_output
30
 
 
55
 
56
  def _default_axmodel_path() -> str:
57
  script_dir = Path(__file__).resolve().parent
58
+ return default_axmodel_path(script_dir)
 
 
 
 
 
 
 
59
 
60
 
61
  def _default_vit_model_path() -> str:
 
63
  resize_h, resize_w, expected_tokens = resolve_resize(DEFAULT_MAX_SOFT_TOKENS)
64
  stem = f"gemma4_vision_h{resize_h}_w{resize_w}_t{expected_tokens}"
65
  candidates = [
66
+ script_dir / f"{stem}.axmodel",
67
  script_dir / "vit_models" / f"{stem}.axmodel",
68
+ script_dir / f"{stem}.onnx",
69
  script_dir / "vit_models" / f"{stem}.onnx",
70
  script_dir.parent / "model_convert" / "compiled_output" / f"{stem}.axmodel",
71
  script_dir.parent / "model_convert" / "vit-models" / f"{stem}.onnx",
 
105
  self.processor = load_processor(hf_model)
106
  self.tokenizer = self.processor.tokenizer
107
  self.config = load_text_runtime_config(hf_model)
108
+ self.embeds = load_text_embeddings(axmodel_path, self.config)
109
  self.axmodel_path = axmodel_path
110
  self.vit_model_path = vit_model_path
111
  self.max_seq_len = max_seq_len
 
342
  parser.add_argument("--hf_model", type=str, default=_default_hf_model(),
343
  help="Path to Gemma 4 tokenizer/config directory")
344
  parser.add_argument("--axmodel_path", type=str, default=_default_axmodel_path(),
345
+ help="Path to the packaged LLM runtime root or legacy axmodel folder")
346
  parser.add_argument("--vit_model_path", type=str, default=_default_vit_model_path(),
347
  help="Path to Gemma 4 vision ONNX model or .axmodel")
348
  parser.add_argument("--port", type=int, default=7860, help="Gradio server port")
infer_axmodel.py CHANGED
@@ -17,6 +17,8 @@ from utils.gemma4_multimodal import replace_image_tokens
17
  from utils.gemma4_multimodal import resolve_resize
18
  from utils.gemma4_multimodal import to_numpy_fp32
19
  from utils.gemma4_per_layer import Gemma4PerLayerInputs
 
 
20
  from utils.infer_func import InferManager
21
  from utils.vision_output import describe_output_shapes
22
  from utils.vision_output import select_vit_output
@@ -36,15 +38,7 @@ def _default_hf_model() -> str:
36
 
37
  def _default_axmodel_path() -> str:
38
  script_dir = Path(__file__).resolve().parent
39
- candidates = [
40
- script_dir / "gemma_4_e2b_it_ax650n_axmodel",
41
- script_dir / "gemma_4_e2b_it_ax650n_w4a16_axmodel",
42
- script_dir / "gemma-4-E2B-it_axmodel",
43
- ]
44
- for candidate in candidates:
45
- if candidate.exists():
46
- return str(candidate)
47
- return str(candidates[0])
48
 
49
 
50
  def _default_vit_model_path() -> str:
@@ -52,7 +46,9 @@ def _default_vit_model_path() -> str:
52
  resize_h, resize_w, expected_tokens = resolve_resize(DEFAULT_MAX_SOFT_TOKENS)
53
  stem = f"gemma4_vision_h{resize_h}_w{resize_w}_t{expected_tokens}"
54
  candidates = [
 
55
  script_dir / "vit_models" / f"{stem}.axmodel",
 
56
  script_dir / "vit_models" / f"{stem}.onnx",
57
  script_dir.parent / "model_convert" / "compiled_output" / f"{stem}.axmodel",
58
  script_dir.parent / "model_convert" / "vit-models" / f"{stem}.onnx",
@@ -111,7 +107,7 @@ if __name__ == "__main__":
111
  parser.add_argument("--hf_model", type=str, default=_default_hf_model(),
112
  help="Path to Gemma 4 tokenizer/config directory")
113
  parser.add_argument("--axmodel_path", type=str, default=_default_axmodel_path(),
114
- help="Path to compiled LLM axmodel folder")
115
  parser.add_argument("--vit_model_path", type=str, default=_default_vit_model_path(),
116
  help="Path to Gemma 4 vision ONNX model or .axmodel")
117
  parser.add_argument("--image_path", type=str, default="",
@@ -135,7 +131,7 @@ if __name__ == "__main__":
135
  args = parser.parse_args()
136
 
137
  config = load_text_runtime_config(args.hf_model)
138
- embeds = np.load(os.path.join(args.axmodel_path, "model.embed_tokens.weight.npy"))
139
  per_layer_helper = None
140
  if int(getattr(config, "hidden_size_per_layer_input", 0) or 0) > 0:
141
  per_layer_helper = Gemma4PerLayerInputs(args.axmodel_path, config)
@@ -147,6 +143,9 @@ if __name__ == "__main__":
147
  print(f"[INFO] Auto-detected max_soft_tokens={detected} from VIT model: {args.vit_model_path}")
148
  args.max_soft_tokens = detected
149
 
 
 
 
150
  mm_token_type_ids = None
151
  prefill_per_layer_inputs = None
152
  if args.image_path:
@@ -233,8 +232,6 @@ if __name__ == "__main__":
233
 
234
  eos_token_id = config.eos_token_id if isinstance(config.eos_token_id, list) else None
235
 
236
- kv_cache_len = int(getattr(config, "kv_cache_len", 2047) or 2047)
237
- imer = InferManager(config, args.axmodel_path, max_seq_len=kv_cache_len, per_layer_helper=per_layer_helper)
238
  token_ids = imer.prefill(
239
  tokenizer,
240
  token_ids,
 
17
  from utils.gemma4_multimodal import resolve_resize
18
  from utils.gemma4_multimodal import to_numpy_fp32
19
  from utils.gemma4_per_layer import Gemma4PerLayerInputs
20
+ from utils.runtime_layout import default_axmodel_path
21
+ from utils.runtime_layout import load_text_embeddings
22
  from utils.infer_func import InferManager
23
  from utils.vision_output import describe_output_shapes
24
  from utils.vision_output import select_vit_output
 
38
 
39
  def _default_axmodel_path() -> str:
40
  script_dir = Path(__file__).resolve().parent
41
+ return default_axmodel_path(script_dir)
 
 
 
 
 
 
 
 
42
 
43
 
44
  def _default_vit_model_path() -> str:
 
46
  resize_h, resize_w, expected_tokens = resolve_resize(DEFAULT_MAX_SOFT_TOKENS)
47
  stem = f"gemma4_vision_h{resize_h}_w{resize_w}_t{expected_tokens}"
48
  candidates = [
49
+ script_dir / f"{stem}.axmodel",
50
  script_dir / "vit_models" / f"{stem}.axmodel",
51
+ script_dir / f"{stem}.onnx",
52
  script_dir / "vit_models" / f"{stem}.onnx",
53
  script_dir.parent / "model_convert" / "compiled_output" / f"{stem}.axmodel",
54
  script_dir.parent / "model_convert" / "vit-models" / f"{stem}.onnx",
 
107
  parser.add_argument("--hf_model", type=str, default=_default_hf_model(),
108
  help="Path to Gemma 4 tokenizer/config directory")
109
  parser.add_argument("--axmodel_path", type=str, default=_default_axmodel_path(),
110
+ help="Path to the packaged LLM runtime root or legacy axmodel folder")
111
  parser.add_argument("--vit_model_path", type=str, default=_default_vit_model_path(),
112
  help="Path to Gemma 4 vision ONNX model or .axmodel")
113
  parser.add_argument("--image_path", type=str, default="",
 
131
  args = parser.parse_args()
132
 
133
  config = load_text_runtime_config(args.hf_model)
134
+ embeds = load_text_embeddings(args.axmodel_path, config)
135
  per_layer_helper = None
136
  if int(getattr(config, "hidden_size_per_layer_input", 0) or 0) > 0:
137
  per_layer_helper = Gemma4PerLayerInputs(args.axmodel_path, config)
 
143
  print(f"[INFO] Auto-detected max_soft_tokens={detected} from VIT model: {args.vit_model_path}")
144
  args.max_soft_tokens = detected
145
 
146
+ kv_cache_len = int(getattr(config, "kv_cache_len", 2047) or 2047)
147
+ imer = InferManager(config, args.axmodel_path, max_seq_len=kv_cache_len, per_layer_helper=per_layer_helper)
148
+
149
  mm_token_type_ids = None
150
  prefill_per_layer_inputs = None
151
  if args.image_path:
 
232
 
233
  eos_token_id = config.eos_token_id if isinstance(config.eos_token_id, list) else None
234
 
 
 
235
  token_ids = imer.prefill(
236
  tokenizer,
237
  token_ids,
gemma_4_e2b_it_ax650n_axmodel/model.embed_tokens.weight.bfloat16.bin → model.embed_tokens.weight.bfloat16.bin RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/embed_tokens_per_layer.weight.npy → model.embed_tokens_per_layer.weight.npy RENAMED
File without changes
gemma_4_e2b_it_ax650n_axmodel/per_layer_model_projection.weight.npy → model.per_layer_model_projection.weight.npy RENAMED
File without changes