array commited on
Commit
f9b2818
·
verified ·
1 Parent(s): d9abb32

Serve correctly on vLLM: fix image_processor_type, pre-fill latent block in chat template, pin vision_config

Browse files

- preprocessor_config.json: image_processor_type Qwen2_5_VLImageProcessor -> Qwen2VLImageProcessor.
The old name is the transformers-fork class and is not in the auto map, so stock
transformers raised 'Unrecognized image processor'. The two classes produce
bit-identical pixel_values.
- chat_template: the add_generation_prompt branch now pre-fills the assistant turn with
<think> + 20 x <|latent_pad|> + </think>. Without it a served model receives zero latent
tokens and reverts to thinking in text (~11x the decode cost). Guarded by
add_generation_prompt, so training and lmms-eval render byte-identically to before.
Override per request with chat_template_kwargs {"num_latents": N}; 0 disables.
- config.json: vision_config written out explicitly rather than relying on transformers
class defaults. auto_map is kept.

chat_template.jinja CHANGED
@@ -4,4 +4,5 @@ You are a helpful assistant.<|im_end|>
4
  {% if message['content'] is string %}{{ message['content'] }}<|im_end|>
5
  {% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
6
  {% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
7
- {% endif %}
 
 
4
  {% if message['content'] is string %}{{ message['content'] }}<|im_end|>
5
  {% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
6
  {% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
7
+ {% set n = num_latents if num_latents is defined else 20 %}{% if n > 0 %}<think>{{ '<|latent_pad|>' * n }}</think>
8
+ {% endif %}{% endif %}
chat_template.json CHANGED
@@ -1,3 +1,3 @@
1
  {
2
- "chat_template": "{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}"
3
- }
 
1
  {
2
+ "chat_template": "{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% set n = num_latents if num_latents is defined else 20 %}{% if n > 0 %}<think>{{ '<|latent_pad|>' * n }}</think>\n{% endif %}{% endif %}"
3
+ }
config.json CHANGED
@@ -48,14 +48,29 @@
48
  "use_sliding_window": false,
49
  "video_token_id": 151656,
50
  "vision_config": {
 
51
  "dtype": "bfloat16",
 
 
 
 
 
 
 
52
  "hidden_size": 1280,
53
  "in_chans": 3,
54
  "initializer_range": 0.02,
 
55
  "model_type": "qwen2_5_vl",
 
 
 
 
56
  "spatial_patch_size": 14,
 
57
  "tokens_per_second": 2,
58
- "torch_dtype": "bfloat16"
 
59
  },
60
  "vision_end_token_id": 151653,
61
  "vision_start_token_id": 151652,
 
48
  "use_sliding_window": false,
49
  "video_token_id": 151656,
50
  "vision_config": {
51
+ "depth": 32,
52
  "dtype": "bfloat16",
53
+ "fullatt_block_indexes": [
54
+ 7,
55
+ 15,
56
+ 23,
57
+ 31
58
+ ],
59
+ "hidden_act": "silu",
60
  "hidden_size": 1280,
61
  "in_chans": 3,
62
  "initializer_range": 0.02,
63
+ "intermediate_size": 3420,
64
  "model_type": "qwen2_5_vl",
65
+ "num_heads": 16,
66
+ "out_hidden_size": 3584,
67
+ "patch_size": 14,
68
+ "spatial_merge_size": 2,
69
  "spatial_patch_size": 14,
70
+ "temporal_patch_size": 2,
71
  "tokens_per_second": 2,
72
+ "torch_dtype": "bfloat16",
73
+ "window_size": 112
74
  },
75
  "vision_end_token_id": 151653,
76
  "vision_start_token_id": 151652,
preprocessor_config.json CHANGED
@@ -8,7 +8,7 @@
8
  0.4578275,
9
  0.40821073
10
  ],
11
- "image_processor_type": "Qwen2_5_VLImageProcessor",
12
  "image_std": [
13
  0.26862954,
14
  0.26130258,
 
8
  0.4578275,
9
  0.40821073
10
  ],
11
+ "image_processor_type": "Qwen2VLImageProcessor",
12
  "image_std": [
13
  0.26862954,
14
  0.26130258,