ZHANGYUXUAN-zR commited on
Commit
52cd0df
·
verified ·
1 Parent(s): ca5d8b3

update configs and tokenizer

Browse files
Files changed (4) hide show
  1. README.md +5 -99
  2. config.json +33 -26
  3. processor_config.json +63 -0
  4. tokenizer_config.json +4 -3
README.md CHANGED
@@ -69,107 +69,13 @@ Compared with model-only inference, the SDK integrates PP-DocLayoutV3 and provid
69
 
70
  Note that the SDK is currently designed for document parsing tasks only. For information extraction tasks, please refer to the following section and run inference directly with the model.
71
 
72
- ### vLLM
73
 
74
- 1. run
75
 
76
- ```bash
77
- pip install -U vllm --extra-index-url https://wheels.vllm.ai/nightly
78
- ```
79
-
80
- or using docker with:
81
- ```
82
- docker pull vllm/vllm-openai:nightly
83
- ```
84
-
85
- 2. run with:
86
-
87
- ```bash
88
- pip install git+https://github.com/huggingface/transformers.git
89
- vllm serve zai-org/GLM-OCR --allowed-local-media-path / --port 8080
90
- ```
91
-
92
- ### SGLang
93
-
94
-
95
- 1. using docker with:
96
-
97
- ```bash
98
- docker pull lmsysorg/sglang:dev
99
- ```
100
-
101
- or build it from source with:
102
-
103
- ```bash
104
- pip install git+https://github.com/sgl-project/sglang.git#subdirectory=python
105
- ```
106
-
107
- 2. run with:
108
-
109
- ```bash
110
- pip install git+https://github.com/huggingface/transformers.git
111
- python -m sglang.launch_server --model zai-org/GLM-OCR --port 8080
112
- ```
113
-
114
- ### Ollama
115
-
116
- 1. Download [Ollama](https://ollama.com/download).
117
- 2. run with:
118
-
119
- ```bash
120
- ollama run glm-ocr
121
- ```
122
-
123
- Ollama will automatically use image file path when an image is dragged into the terminal:
124
-
125
- ```bash
126
- ollama run glm-ocr Text Recognition: ./image.png
127
- ```
128
-
129
- ### Transformers
130
-
131
- ```
132
- pip install git+https://github.com/huggingface/transformers.git
133
- ```
134
-
135
- ```python
136
- from transformers import AutoProcessor, AutoModelForImageTextToText
137
- import torch
138
-
139
- MODEL_PATH = "zai-org/GLM-OCR"
140
- messages = [
141
- {
142
- "role": "user",
143
- "content": [
144
- {
145
- "type": "image",
146
- "url": "test_image.png"
147
- },
148
- {
149
- "type": "text",
150
- "text": "Text Recognition:"
151
- }
152
- ],
153
- }
154
- ]
155
- processor = AutoProcessor.from_pretrained(MODEL_PATH)
156
- model = AutoModelForImageTextToText.from_pretrained(
157
- pretrained_model_name_or_path=MODEL_PATH,
158
- torch_dtype="auto",
159
- device_map="auto",
160
- )
161
- inputs = processor.apply_chat_template(
162
- messages,
163
- tokenize=True,
164
- add_generation_prompt=True,
165
- return_dict=True,
166
- return_tensors="pt"
167
- ).to(model.device)
168
- inputs.pop("token_type_ids", None)
169
- generated_ids = model.generate(**inputs, max_new_tokens=8192)
170
- output_text = processor.decode(generated_ids[0][inputs["input_ids"].shape[1]:], skip_special_tokens=False)
171
- print(output_text)
172
- ```
173
 
174
  ### Prompt Limited
175
 
 
69
 
70
  Note that the SDK is currently designed for document parsing tasks only. For information extraction tasks, please refer to the following section and run inference directly with the model.
71
 
72
+ ### Serve GLM-OCR Locally
73
 
74
+ GLM-OCR supports deployment with the following frameworks. Feel free to try them out:
75
 
76
+ - [SGLang](https://github.com/sgl-project/sglang) — see [cookbook](https://cookbook.sglang.io/autoregressive/GLM/GLM-OCR)
77
+ - [vLLM](https://github.com/vllm-project/vllm) — see [recipes](https://recipes.vllm.ai/zai-org/GLM-OCR)
78
+ - [Transformers](https://github.com/huggingface/transformers) — see [transformers docs](https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/glm_ocr.md)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
79
 
80
  ### Prompt Limited
81
 
config.json CHANGED
@@ -2,64 +2,71 @@
2
  "architectures": [
3
  "GlmOcrForConditionalGeneration"
4
  ],
 
 
 
5
  "model_type": "glm_ocr",
6
  "text_config": {
7
- "model_type": "glm_ocr_text",
8
- "pad_token_id": 59246,
9
- "vocab_size": 59392,
10
  "eos_token_id": [
11
  59246,
12
  59253
13
  ],
14
- "attention_bias": false,
15
- "attention_dropout": 0.0,
16
  "head_dim": 128,
17
  "hidden_act": "silu",
18
  "hidden_size": 1536,
19
  "initializer_range": 0.02,
20
  "intermediate_size": 4608,
21
  "max_position_embeddings": 131072,
 
22
  "num_attention_heads": 16,
23
  "num_hidden_layers": 16,
24
- "num_nextn_predict_layers": 1,
25
  "num_key_value_heads": 8,
 
 
26
  "rms_norm_eps": 1e-05,
27
- "dtype": "bfloat16",
28
  "rope_parameters": {
29
- "rope_type": "default",
30
  "mrope_section": [
31
  16,
32
  24,
33
  24
34
  ],
35
- "partial_rotary_factor": 1.0,
36
- "rope_theta": 10000
 
37
  },
38
  "tie_word_embeddings": false,
39
- "use_cache": true
 
40
  },
 
 
 
 
 
41
  "vision_config": {
42
- "model_type": "glm_ocr_vision",
43
- "hidden_size": 1024,
44
- "depth": 24,
45
- "num_heads": 16,
46
  "attention_bias": true,
47
- "intermediate_size": 4096,
 
48
  "hidden_act": "silu",
49
  "hidden_dropout_prob": 0.0,
50
- "initializer_range": 0.02,
51
  "image_size": 336,
52
- "patch_size": 14,
 
 
 
 
53
  "out_hidden_size": 1536,
 
54
  "rms_norm_eps": 1e-05,
 
 
 
 
55
  "spatial_merge_size": 2,
56
  "temporal_patch_size": 2
57
- },
58
- "image_start_token_id": 59256,
59
- "image_end_token_id": 59257,
60
- "video_start_token_id": 59258,
61
- "video_end_token_id": 59259,
62
- "image_token_id": 59280,
63
- "video_token_id": 59281,
64
- "transformers_version": "5.0.1dev0"
65
  }
 
2
  "architectures": [
3
  "GlmOcrForConditionalGeneration"
4
  ],
5
+ "image_end_token_id": 59257,
6
+ "image_start_token_id": 59256,
7
+ "image_token_id": 59280,
8
  "model_type": "glm_ocr",
9
  "text_config": {
10
+ "attention_bias": false,
11
+ "attention_dropout": 0.0,
12
+ "dtype": "bfloat16",
13
  "eos_token_id": [
14
  59246,
15
  59253
16
  ],
 
 
17
  "head_dim": 128,
18
  "hidden_act": "silu",
19
  "hidden_size": 1536,
20
  "initializer_range": 0.02,
21
  "intermediate_size": 4608,
22
  "max_position_embeddings": 131072,
23
+ "model_type": "glm_ocr_text",
24
  "num_attention_heads": 16,
25
  "num_hidden_layers": 16,
 
26
  "num_key_value_heads": 8,
27
+ "num_nextn_predict_layers": 1,
28
+ "pad_token_id": 59246,
29
  "rms_norm_eps": 1e-05,
 
30
  "rope_parameters": {
 
31
  "mrope_section": [
32
  16,
33
  24,
34
  24
35
  ],
36
+ "partial_rotary_factor": 1.0,
37
+ "rope_theta": 10000,
38
+ "rope_type": "default"
39
  },
40
  "tie_word_embeddings": false,
41
+ "use_cache": true,
42
+ "vocab_size": 59392
43
  },
44
+ "tie_word_embeddings": false,
45
+ "transformers_version": "5.17.0",
46
+ "video_end_token_id": 59259,
47
+ "video_start_token_id": 59258,
48
+ "video_token_id": 59281,
49
  "vision_config": {
 
 
 
 
50
  "attention_bias": true,
51
+ "attention_dropout": 0.0,
52
+ "depth": 24,
53
  "hidden_act": "silu",
54
  "hidden_dropout_prob": 0.0,
55
+ "hidden_size": 1024,
56
  "image_size": 336,
57
+ "in_channels": 3,
58
+ "initializer_range": 0.02,
59
+ "intermediate_size": 4096,
60
+ "model_type": "glm_ocr_vision",
61
+ "num_heads": 16,
62
  "out_hidden_size": 1536,
63
+ "patch_size": 14,
64
  "rms_norm_eps": 1e-05,
65
+ "rope_parameters": {
66
+ "rope_theta": 10000.0,
67
+ "rope_type": "axial"
68
+ },
69
  "spatial_merge_size": 2,
70
  "temporal_patch_size": 2
71
+ }
 
 
 
 
 
 
 
72
  }
processor_config.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image_processor": {
3
+ "do_convert_rgb": true,
4
+ "do_normalize": true,
5
+ "do_rescale": true,
6
+ "do_resize": true,
7
+ "image_mean": [
8
+ 0.48145466,
9
+ 0.4578275,
10
+ 0.40821073
11
+ ],
12
+ "image_processor_type": "Glm46VImageProcessor",
13
+ "image_std": [
14
+ 0.26862954,
15
+ 0.26130258,
16
+ 0.27577711
17
+ ],
18
+ "merge_size": 2,
19
+ "patch_size": 14,
20
+ "resample": 3,
21
+ "rescale_factor": 0.00392156862745098,
22
+ "size": {
23
+ "longest_edge": 9633792,
24
+ "shortest_edge": 12544
25
+ },
26
+ "temporal_patch_size": 2
27
+ },
28
+ "processor_class": "Glm46VProcessor",
29
+ "video_processor": {
30
+ "do_convert_rgb": true,
31
+ "do_normalize": true,
32
+ "do_rescale": true,
33
+ "do_resize": true,
34
+ "do_sample_frames": true,
35
+ "fps": 2,
36
+ "image_mean": [
37
+ 0.48145466,
38
+ 0.4578275,
39
+ 0.40821073
40
+ ],
41
+ "image_std": [
42
+ 0.26862954,
43
+ 0.26130258,
44
+ 0.27577711
45
+ ],
46
+ "max_duration": 300,
47
+ "max_image_size": {
48
+ "longest_edge": 47040000
49
+ },
50
+ "merge_size": 2,
51
+ "num_frames": 16,
52
+ "patch_size": 14,
53
+ "resample": 3,
54
+ "rescale_factor": 0.00392156862745098,
55
+ "return_metadata": false,
56
+ "size": {
57
+ "longest_edge": 9633792,
58
+ "shortest_edge": 12544
59
+ },
60
+ "temporal_patch_size": 2,
61
+ "video_processor_type": "Glm46VVideoProcessor"
62
+ }
63
+ }
tokenizer_config.json CHANGED
@@ -36,12 +36,13 @@
36
  "</arg_value>",
37
  "/nothink",
38
  "<|begin_of_box|>",
39
- "<|end_of_box|>",
40
- "<|image|>",
41
- "<|video|>"
42
  ],
43
  "is_local": true,
44
  "model_max_length": 655380,
 
 
 
45
  "pad_token": "<|endoftext|>",
46
  "padding_side": "left",
47
  "processor_class": "Glm46VProcessor",
 
36
  "</arg_value>",
37
  "/nothink",
38
  "<|begin_of_box|>",
39
+ "<|end_of_box|>"
 
 
40
  ],
41
  "is_local": true,
42
  "model_max_length": 655380,
43
+ "clean_up_tokenization_spaces": false,
44
+ "do_lower_case": false,
45
+ "model_specific_special_tokens": {},
46
  "pad_token": "<|endoftext|>",
47
  "padding_side": "left",
48
  "processor_class": "Glm46VProcessor",