tomaxe commited on Apr 14

Commit

6d8f668

verified ·

1 Parent(s): 8eaca0c

Upload folder using huggingface_hub

Browse files

Files changed (43) hide show

.gitattributes +1 -0
README.md +196 -0
added_tokens.json +16 -0
chat_template.json +3 -0
config.json +52 -0
merges.txt +0 -0
model-00001-of-00031.safetensors +3 -0
model-00002-of-00031.safetensors +3 -0
model-00003-of-00031.safetensors +3 -0
model-00004-of-00031.safetensors +3 -0
model-00005-of-00031.safetensors +3 -0
model-00006-of-00031.safetensors +3 -0
model-00007-of-00031.safetensors +3 -0
model-00008-of-00031.safetensors +3 -0
model-00009-of-00031.safetensors +3 -0
model-00010-of-00031.safetensors +3 -0
model-00011-of-00031.safetensors +3 -0
model-00012-of-00031.safetensors +3 -0
model-00013-of-00031.safetensors +3 -0
model-00014-of-00031.safetensors +3 -0
model-00015-of-00031.safetensors +3 -0
model-00016-of-00031.safetensors +3 -0
model-00017-of-00031.safetensors +3 -0
model-00018-of-00031.safetensors +3 -0
model-00019-of-00031.safetensors +3 -0
model-00020-of-00031.safetensors +3 -0
model-00021-of-00031.safetensors +3 -0
model-00022-of-00031.safetensors +3 -0
model-00023-of-00031.safetensors +3 -0
model-00024-of-00031.safetensors +3 -0
model-00025-of-00031.safetensors +3 -0
model-00026-of-00031.safetensors +3 -0
model-00027-of-00031.safetensors +3 -0
model-00028-of-00031.safetensors +3 -0
model-00029-of-00031.safetensors +3 -0
model-00030-of-00031.safetensors +3 -0
model-00031-of-00031.safetensors +3 -0
model.safetensors.index.json +0 -0
preprocessor_config.json +29 -0
special_tokens_map.json +31 -0
tokenizer.json +3 -0
tokenizer_config.json +145 -0
vocab.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,196 @@

+---
+license: apache-2.0
+language:
+- en
+pipeline_tag: image-text-to-text
+tags:
+- multimodal
+- gui
+library_name: transformers
+---
+# UI-TARS-72B-DPO
+[UI-TARS-2B-SFT](https://huggingface.co/bytedance-research/UI-TARS-2B-SFT) &nbsp;|&nbsp;
+[UI-TARS-7B-SFT](https://huggingface.co/bytedance-research/UI-TARS-7B-SFT) &nbsp;|&nbsp;
+[**UI-TARS-7B-DPO**](https://huggingface.co/bytedance-research/UI-TARS-7B-DPO)(Recommended) &nbsp;|&nbsp;
+[UI-TARS-72B-SFT](https://huggingface.co/bytedance-research/UI-TARS-72B-SFT) &nbsp;|&nbsp;
+[**UI-TARS-72B-DPO**](https://huggingface.co/bytedance-research/UI-TARS-72B-DPO)(Recommended)
+## Introduction
+UI-TARS is a next-generation native GUI agent model designed to interact seamlessly with graphical user interfaces (GUIs) using human-like perception, reasoning, and action capabilities. Unlike traditional modular frameworks, UI-TARS integrates all key components—perception, reasoning, grounding, and memory—within a single vision-language model (VLM), enabling end-to-end task automation without predefined workflows or manual rules.
+<!-- ![Local Image](figures/UI-TARS.png) -->
+<p align="center">
+    <img src="https://github.com/bytedance/UI-TARS/blob/main/figures/UI-TARS-vs-Previous-SOTA.png?raw=true" width="90%"/>
+<p>
+<p align="center">
+    <img src="https://github.com/bytedance/UI-TARS/blob/main/figures/UI-TARS.png?raw=true" width="90%"/>
+<p>
+<!-- ![Local Image](figures/UI-TARS-vs-Previous-SOTA.png) -->
+This repository contains the model for the paper [UI-TARS: Pioneering Automated GUI Interaction with Native Agents](https://huggingface.co/papers/2501.12326).
+Code: https://github.com/bytedance/UI-TARS
+## Performance
+**Perception Capabilty Evaluation**
+| Model                     | VisualWebBench | WebSRC  | SQAshort |
+|---------------------------|---------------|---------|----------|
+| Qwen2-VL-7B              | 73.3          | 81.8    | 84.9     |
+| Qwen-VL-Max              | 74.1          | 91.1    | 78.6     |
+| Gemini-1.5-Pro           | 75.4          | 88.9    | 82.2     |
+| UIX-Qwen2-7B             | 75.9          | 82.9    | 78.8     |
+| Claude-3.5-Sonnet        | 78.2          | 90.4    | 83.1     |
+| GPT-4o                   | 78.5          | 87.7    | 82.3     |
+| **UI-TARS-2B**          | 72.9          | 89.2    | 86.4     |
+| **UI-TARS-7B**          | 79.7          | **93.6** | 87.7     |
+| **UI-TARS-72B**         | **82.8**      | 89.3    | **88.6** |
+**Grounding Capability Evaluation**
+- **ScreenSpot Pro**
+| Agent Model              | Dev-Text | Dev-Icon | Dev-Avg | Creative-Text | Creative-Icon | Creative-Avg | CAD-Text | CAD-Icon | CAD-Avg | Scientific-Text | Scientific-Icon | Scientific-Avg | Office-Text | Office-Icon | Office-Avg | OS-Text | OS-Icon | OS-Avg | Avg-Text | Avg-Icon | Avg |
+|--------------------------|----------|----------|----------|--------------|--------------|--------------|---------|---------|---------|---------------|---------------|---------------|------------|------------|------------|--------|--------|--------|---------|---------|------|
+| QwenVL-7B               | 0.0      | 0.0      | 0.0      | 0.0          | 0.0          | 0.0          | 0.0     | 0.0     | 0.0     | 0.7           | 0.0           | 0.4           | 0.0        | 0.0        | 0.0        | 0.0    | 0.0    | 0.0    | 0.1     | 0.0     | **0.1**  |
+| GPT-4o                  | 1.3      | 0.0      | 0.7      | 1.0          | 0.0          | 0.6          | 2.0     | 0.0     | 1.5     | 2.1           | 0.0           | 1.2           | 1.1        | 0.0        | 0.9        | 0.0    | 0.0    | 0.0    | 1.3     | 0.0     | **0.8**  |
+| SeeClick                | 0.6      | 0.0      | 0.3      | 1.0          | 0.0          | 0.6          | 2.5     | 0.0     | 1.9     | 3.5           | 0.0           | 2.0           | 1.1        | 0.0        | 0.9        | 2.8    | 0.0    | 1.5    | 1.8     | 0.0     | **1.1**  |
+| Qwen2-VL-7B             | 2.6      | 0.0      | 1.3      | 1.5          | 0.0          | 0.9          | 0.5     | 0.0     | 0.4     | 6.3           | 0.0           | 3.5           | 3.4        | 1.9        | 3.0        | 0.9    | 0.0    | 0.5    | 2.5     | 0.2     | **1.6**  |
+| OS-Atlas-4B            | 7.1      | 0.0      | 3.7      | 3.0          | 1.4          | 2.3          | 2.0     | 0.0     | 1.5     | 9.0           | 5.5           | 7.5           | 5.1        | 3.8        | 4.8        | 5.6    | 0.0    | 3.1    | 5.0     | 1.7     | **3.7**  |
+| ShowUI-2B              | 16.9     | 1.4      | 9.4      | 9.1          | 0.0          | 5.3          | 2.5     | 0.0     | 1.9     | 13.2          | 7.3           | 10.6          | 15.3       | 7.5        | 13.5       | 10.3   | 2.2    | 6.6    | 10.8    | 2.6     | **7.7**  |
+| CogAgent-18B           | 14.9     | 0.7      | 8.0      | 9.6          | 0.0          | 5.6          | 7.1     | 3.1     | 6.1     | 22.2          | 1.8           | 13.4          | 13.0       | 0.0        | 10.0       | 5.6    | 0.0    | 3.1    | 12.0    | 0.8     | **7.7**  |
+| Aria-UI                | 16.2     | 0.0      | 8.4      | 23.7         | 2.1          | 14.7         | 7.6     | 1.6     | 6.1     | 27.1          | 6.4           | 18.1          | 20.3       | 1.9        | 16.1       | 4.7    | 0.0    | 2.6    | 17.1    | 2.0     | **11.3**  |
+| UGround-7B             | 26.6     | 2.1      | 14.7     | 27.3         | 2.8          | 17.0         | 14.2    | 1.6     | 11.1    | 31.9          | 2.7           | 19.3          | 31.6       | 11.3       | 27.0       | 17.8   | 0.0    | 9.7    | 25.0    | 2.8     | **16.5**  |
+| Claude Computer Use      | 22.0  | 3.9   | 12.6  | 25.9  | 3.4   | 16.8  | 14.5  | 3.7   | 11.9  | 33.9  | 15.8  | 25.8  | 30.1  | 16.3  | 26.9  | 11.0  | 4.5   | 8.1   | 23.4  | 7.1  | **17.1**  |
+| OS-Atlas-7B              | 33.1  | 1.4   | 17.7  | 28.8  | 2.8   | 17.9  | 12.2  | 4.7   | 10.3  | 37.5  | 7.3   | 24.4  | 33.9  | 5.7   | 27.4  | 27.1  | 4.5   | 16.8  | 28.1  | 4.0  | **18.9**  |
+| UGround-V1-7B            | -     | -     | 35.5  | -     | -     | 27.8  | -     | -     | 13.5  | -     | -     | 38.8  | -     | -     | 48.8  | -     | -     | 26.1  | -     | -    | **31.1**  |
+| **UI-TARS-2B**        | 47.4     | 4.1      | 26.4     | 42.9         | 6.3          | 27.6         | 17.8    | 4.7     | 14.6    | 56.9          | 17.3          | 39.8          | 50.3       | 17.0       | 42.6       | 21.5   | 5.6    | 14.3   | 39.6    | 8.4     | **27.7**  |
+| **UI-TARS-7B**        | 58.4     | 12.4     | 36.1     | 50.0         | 9.1          | 32.8         | **20.8**| 9.4     | **18.0**| 63.9          | **31.8**      | **50.0**      | **63.3**   | 20.8       | 53.5       | 30.8   | **16.9**| 24.5   | 47.8    | 16.2    | **35.7**  |
+| **UI-TARS-72B**       | **63.0** | **17.3** | **40.8** | **57.1**     | **15.4**     | **39.6**     | 18.8    | **12.5**| 17.2    | **64.6**      | 20.9          | 45.7          | **63.3**   | **26.4**   | **54.8**   | **42.1**| 15.7    | **30.1**| **50.9**| **17.5**| **38.1**  |
+- **ScreenSpot**
+| Method |  Mobile-Text | Mobile-Icon/Widget | Desktop-Text | Desktop-Icon/Widget | Web-Text | Web-Icon/Widget | Avg |
+|--------|-------------|-------------|-------------|-------------|-------------|---------|---------|
+| **Agent Framework**  | | | | | | | |
+| GPT-4 (SeeClick) |  76.6 | 55.5 | 68.0 | 28.6 | 40.9 | 23.3 | **48.8** |
+| GPT-4 (OmniParser)  | 93.9 | 57.0 | 91.3 | 63.6 | 81.3 | 51.0 | **73.0** |
+| GPT-4 (UGround-7B)  | 90.1 | 70.3 | 87.1 | 55.7 | 85.7 | 64.6 | **75.6** |
+| GPT-4o (SeeClick)  | 81.0 | 59.8 | 69.6 | 33.6 | 43.9 | 26.2 | **52.3** |
+| GPT-4o (UGround-7B)  | 93.4 | 76.9 | 92.8 | 67.9 | 88.7 | 68.9 | **81.4** |
+| **Agent Model**   | | | | | | | |
+| GPT-4  | 22.6 | 24.5 | 20.2 | 11.8 | 9.2 | 8.8 | **16.2** |
+| GPT-4o  | 20.2 | 24.9 | 21.1 | 23.6 | 12.2 | 7.8 | **18.3** |
+| CogAgent  | 67.0 | 24.0 | 74.2 | 20.0 | 70.4 | 28.6 | **47.4** |
+| SeeClick  | 78.0 | 52.0 | 72.2 | 30.0 | 55.7 | 32.5 | **53.4** |
+| Qwen2-VL  | 75.5 | 60.7 | 76.3 | 54.3 | 35.2 | 25.7 | **55.3** |
+| UGround-7B  | 82.8 | 60.3 | 82.5 | 63.6 | 80.4 | 70.4 | **73.3** |
+| Aguvis-G-7B  | 88.3 | 78.2 | 88.1 | 70.7 | 85.7 | 74.8 | **81.8** |
+| OS-Atlas-7B | 93.0 | 72.9 | 91.8 | 62.9 | 90.9 | 74.3 | **82.5** |
+| Claude Computer Use  | - | - | - | - | - | - | **83.0** |
+| Gemini 2.0 (Project Mariner)  | - | - | - | - | - | - | **84.0** |
+| Aguvis-7B  | **95.6** | 77.7 | 93.8 | 67.1 | 88.3 | 75.2 | **84.4** |
+| Aguvis-72B  | 94.5 | **85.2** | 95.4 | 77.9 | **91.3** | **85.9** | **89.2** |
+| **Our Model**   | | | | | | | |
+| **UI-TARS-2B**  | 93.0 | 75.5 | 90.7 | 68.6 | 84.3 | 74.8 | **82.3** |
+| **UI-TARS-7B**  | 94.5 | **85.2** | **95.9** | 85.7 | 90.0 | 83.5 | **89.5** |
+| **UI-TARS-72B**  | 94.9 | 82.5 | 89.7 | **88.6** | 88.7 | 85.0 | **88.4** |
+- **ScreenSpot v2**
+| Method |  Mobile-Text | Mobile-Icon/Widget | Desktop-Text | Desktop-Icon/Widget | Web-Text | Web-Icon/Widget | Avg |
+|--------|-------------|-------------|-------------|-------------|-------------|---------|---------|
+| **Agent Framework**  | | | | | | | |
+| GPT-4o (SeeClick)  | 85.2 | 58.8 | 79.9 | 37.1 | 72.7 | 30.1 | **63.6** |
+| GPT-4o (OS-Atlas-4B)  | 95.5 | 75.8 | 79.4 | 49.3 | 90.2 | 66.5 | **79.1** |
+| GPT-4o (OS-Atlas-7B)  | 96.2 | 83.4 | 89.7 | 69.3 | **94.0** | 79.8 | **87.1** |
+| **Agent Model**  | | | | | | | |
+| SeeClick  | 78.4 | 50.7 | 70.1 | 29.3 | 55.2 | 32.5 | **55.1** |
+| OS-Atlas-4B  | 87.2 | 59.7 | 72.7 | 46.4 | 85.9 | 63.1 | **71.9** |
+| OS-Atlas-7B  | 95.2 | 75.8 | 90.7 | 63.6 | 90.6 | 77.3 | **84.1** |
+| **Our Model**  | | | | | | | |
+| **UI-TARS-2B**  | 95.2 | 79.1 | 90.7 | 68.6 | 87.2 | 78.3 | **84.7** |
+| **UI-TARS-7B** | **96.9** | **89.1** | **95.4** | 85.0 | 93.6 | 85.2 | **91.6** |
+| **UI-TARS-72B**  | 94.8 | 86.3 | 91.2 | **87.9** | 91.5 | **87.7** | **90.3** |
+**Offline Agent Capability Evaluation**
+- **Multimodal Mind2Web**
+| Method |  Cross-Task Ele.Acc | Cross-Task Op.F1 | Cross-Task Step SR | Cross-Website Ele.Acc | Cross-Website Op.F1 | Cross-Website Step SR | Cross-Domain Ele.Acc | Cross-Domain Op.F1 | Cross-Domain Step SR |
+|--------|----------------------|-------------------|--------------------|----------------------|--------------------|-------------------|--------------------|-------------------|-------------------|
+| **Agent Framework**  | | | | | | | | | |
+| GPT-4o (SeeClick)  | 32.1 | - | - | 33.1 | - | - | 33.5 | - | - |
+| GPT-4o (UGround)  | 47.7 | - | - | 46.0 | - | - | 46.6 | - | - |
+| GPT-4o (Aria-UI)  | 57.6 | - | - | 57.7 | - | - | 61.4 | - | - |
+| GPT-4V (OmniParser)  | 42.4 | 87.6 | 39.4 | 41.0 | 84.8 | 36.5 | 45.5 | 85.7 | 42.0 |
+| **Agent Model** |  | | | | | | | | |
+| GPT-4o  | 5.7 | 77.2 | 4.3 | 5.7 | 79.0 | 3.9 | 5.5 | 86.4 | 4.5 |
+| GPT-4 (SOM)  | 29.6 | - | 20.3 | 20.1 | - | 13.9 | 27.0 | - | 23.7 |
+| GPT-3.5 (Text-only)  | 19.4 | 59.2 | 16.8 | 14.9 | 56.5 | 14.1 | 25.2 | 57.9 | 24.1 |
+| GPT-4 (Text-only)  | 40.8 | 63.1 | 32.3 | 30.2 | 61.0 | 27.0 | 35.4 | 61.9 | 29.7 |
+| Claude  | 62.7 | 84.7 | 53.5 | 59.5 | 79.6 | 47.7 | 64.5 | 85.4 | 56.4 |
+| Aguvis-7B  | 64.2 | 89.8 | 60.4 | 60.7 | 88.1 | 54.6 | 60.4 | 89.2 | 56.6 |
+| CogAgent  | - | - | 62.3 | - | - | 54.0 | - | - | 59.4 |
+| Aguvis-72B  | 69.5 | 90.8 | 64.0 | 62.6 | 88.6 | 56.5 | 63.5 | 88.5 | 58.2 |
+| **Our Model**  | | | | | | | | | |
+| **UI-TARS-2B**  | 62.3 | 90.0 | 56.3 | 58.5 | 87.2 | 50.8 | 58.8 | 89.6 | 52.3 |
+| **UI-TARS-7B**  | 73.1 | 92.2 | 67.1 | 68.2 | 90.9 | 61.7 | 66.6 | 90.9 | 60.5 |
+| **UI-TARS-72B**  | **74.7** | **92.5** | **68.6** | **72.4** | **91.2** | **63.5** | **68.9** | **91.8** | **62.1** |
+- **Android Control and GUI Odyssey**
+| Agent Models        | AndroidControl-Low Type | AndroidControl-Low Grounding | AndroidControl-Low SR | AndroidControl-High Type | AndroidControl-High Grounding | AndroidControl-High SR | GUIOdyssey Type | GUIOdyssey Grounding | GUIOdyssey SR |
+|---------------------|----------------------|----------------------|----------------|----------------------|----------------------|----------------|----------------|----------------|----------------|
+| Claude             | 74.3                 | 0.0                  | 19.4           | 63.7                 | 0.0                  | 12.5           | 60.9           | 0.0            | 3.1            |
+| GPT-4o             | 74.3                 | 0.0                  | 19.4           | 66.3                 | 0.0                  | 20.8           | 34.3           | 0.0            | 3.3            |
+| SeeClick           | 93.0                 | 73.4                 | 75.0           | 82.9                 | 62.9                 | 59.1           | 71.0           | 52.4           | 53.9           |
+| InternVL-2-4B      | 90.9                 | 84.1                 | 80.1           | 84.1                 | 72.7                 | 66.7           | 82.1           | 55.5           | 51.5           |
+| Qwen2-VL-7B       | 91.9                 | 86.5                 | 82.6           | 83.8                 | 77.7                 | 69.7           | 83.5           | 65.9           | 60.2           |
+| Aria-UI           | --                   | 87.7                 | 67.3           | --                   | 43.2                 | 10.2           | --             | 86.8           | 36.5           |
+| OS-Atlas-4B       | 91.9                 | 83.8                 | 80.6           | 84.7                 | 73.8                 | 67.5           | 83.5           | 61.4           | 56.4           |
+| OS-Atlas-7B       | 93.6                 | 88.0                 | 85.2           | 85.2                 | 78.5                 | 71.2           | 84.5           | 67.8           | 62.0           |
+| Aguvis-7B         | --                   | --                   | 80.5           | --                   | --                   | 61.5           | --             | --             | --             |
+| Aguvis-72B        | --                   | --                   | 84.4           | --                   | --                   | 66.4           | --             | --             | --             |
+| **UI-TARS-2B**   | **98.1**             | 87.3                 | 89.3           | 81.2                 | 78.4                 | 68.9           | 93.9           | 86.8           | 83.4           |
+| **UI-TARS-7B**   | 98.0                 | 89.3                 | 90.8           | 83.7                 | 80.5                 | 72.5           | 94.6           | 90.1           | 87.0           |
+| **UI-TARS-72B**  | **98.1**             | **89.9**             | **91.3**       | **85.2**             | **81.5**             | **74.7**       | **95.4**       | **91.4**       | **88.6**       |
+**Online Agent Capability Evaluation**
+| Method |  OSWorld (Online) | AndroidWorld (Online) |
+|--------|-------------------|------------------|
+| **Agent Framework**  | | |
+| GPT-4o (UGround)  | - | 32.8 |
+| GPT-4o (Aria-UI)  | 15.2 | 44.8 |
+| GPT-4o (Aguvis-7B)  | 14.8 | 37.1 |
+| GPT-4o (Aguvis-72B)  | 17.0 | - |
+| GPT-4o (OS-Atlas-7B)  | 14.6 | - |
+| **Agent Model**  | | |
+| GPT-4o  | 5.0 | 34.5 (SoM) |
+| Gemini-Pro-1.5  | 5.4 | 22.8 (SoM) |
+| Aguvis-72B  | 10.3 | 26.1 |
+| Claude Computer-Use  | 14.9 (15 steps) | 27.9 |
+| Claude Computer-Use  | 22.0 (50 steps) | - |
+| **Our Model**  | | |
+| **UI-TARS-7B-SFT**  | 17.7 (15 steps) | 33.0 |
+| **UI-TARS-7B-DPO**  | 18.7 (15 steps) | - |
+| **UI-TARS-72B-SFT**  | 18.8 (15 steps) | **46.6** |
+| **UI-TARS-72B-DPO**  | **22.7** (15 steps) | - |
+| **UI-TARS-72B-DPO**  | **24.6** (50 steps) | - |
+## Citation
+If you find our paper and model useful in your research, feel free to give us a cite.
+```BibTeX
+@article{qin2025ui,
+  title={UI-TARS: Pioneering Automated GUI Interaction with Native Agents},
+  author={Qin, Yujia and Ye, Yining and Fang, Junjie and Wang, Haoming and Liang, Shihao and Tian, Shizuo and Zhang, Junda and Li, Jiahao and Li, Yunxin and Huang, Shijue and others},
+  journal={arXiv preprint arXiv:2501.12326},
+  year={2025}
+}
+```

added_tokens.json ADDED Viewed

	@@ -0,0 +1,16 @@

+{
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

chat_template.json ADDED Viewed

	@@ -0,0 +1,3 @@

+{
+  "chat_template": "{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}"
+}

config.json ADDED Viewed

	@@ -0,0 +1,52 @@

+{
+  "architectures": [
+    "Qwen2VLForConditionalGeneration"
+  ],
+  "attention_dropout": 0.0,
+  "bos_token_id": 151643,
+  "eos_token_id": 151645,
+  "vision_start_token_id": 151652,
+  "vision_end_token_id": 151653,
+  "vision_token_id": 151654,
+  "image_token_id": 151655,
+  "video_token_id": 151656,
+  "hidden_act": "silu",
+  "hidden_size": 8192,
+  "initializer_range": 0.02,
+  "intermediate_size": 29568,
+  "max_position_embeddings": 32768,
+  "max_window_layers": 80,
+  "model_type": "qwen2_vl",
+  "num_attention_heads": 64,
+  "num_hidden_layers": 80,
+  "num_key_value_heads": 8,
+  "rms_norm_eps": 1e-06,
+  "rope_theta": 1000000.0,
+  "sliding_window": 32768,
+  "tie_word_embeddings": false,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.41.2",
+  "use_cache": true,
+  "use_sliding_window": false,
+  "vision_config": {
+    "depth": 32,
+    "embed_dim": 1280,
+    "mlp_ratio": 4,
+    "num_heads": 16,
+    "in_chans": 3,
+    "hidden_size": 8192,
+    "patch_size": 14,
+    "spatial_merge_size": 2,
+    "spatial_patch_size": 14,
+    "temporal_patch_size": 2
+  },
+  "rope_scaling": {
+    "type": "mrope",
+    "mrope_section": [
+      16,
+      24,
+      24
+    ]
+  },
+  "vocab_size": 152064
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model-00001-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0c9fc09de938e9b6627864b6b3be7905e9c4849247d831ed199144a149722531
+size 4676624560

model-00002-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a22a078114f4d7bfa6fc91a1002219881ea502037cb794f3777efe4bd16a88f9
+size 4781670320

model-00003-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:97c2ee1a042cc81bb9e41f2e654f7837e82bd77fc21490274766ebd2d31381d6
+size 4964101384

model-00004-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e7b6001fddafc7333ed75e22517926f44ff1d453b8bae467de67749caa542351
+size 4781637328

model-00005-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4d25d7e4e4dce1ea2a60666a9c81405a3913a5a472bf08fbb5ae45abbc942ee5
+size 4781670344

model-00006-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ccc9fd65861c3d7db80ac967cb0df99a34367c7b486062bac0e3b7d50b1aba39
+size 4781670360

model-00007-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7ed86ce42b9adb688e851bb562ff02ca5192912227ba0faa8b11ff8e2ee69ff4
+size 4964101416

model-00008-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cd4e57407aaa2f5d367c9595f4a7d57d644b1094959ddc81bce930a52df5a8ff
+size 4781637360

model-00009-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:75c4aef80740a12daa79b7f5c94ae043d29ec480f3b83a7dab93983a9c512f50
+size 4781670360

model-00010-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f935d7e9d84311d34a635534d542c09d1b00372444937a851f49c01d5226d7a6
+size 4781670360

model-00011-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:88e6a2af9c8ae08069213db4033d5563ab47fcbf01c98a658c82778c240516aa
+size 4964101416

model-00012-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8b5b3036928363ae058af68e61888830e906126f432c77042ce6a29f83b821f0
+size 4781637360

model-00013-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4ec0599dbc92b989b0593c8331be2549833f8b5a4762b5a9bef7548ca8c77ad2
+size 4781670360

model-00014-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e1deb9bf5576ed05a3ec6666f2e48bcf4ba171dfea6b98d314100d44e4232957
+size 4781670360

model-00015-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:92db5b078ae8eb99f0bebd9797c6f3d7295aaeb0213ae60551da15ffc8b9e296
+size 4964101416

model-00016-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9c437ca3e83616d14399f06d77ba922b4c3f3b3678f218f470a31f124d07ddd2
+size 4781637360

model-00017-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d37246f18b932371067f43dad919ffd56540fb0f343b748870efbeca857a6fbf
+size 4781670360

model-00018-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2f86f1c4fd0b9d0c6d3e0b939f924eb2d75fc0ddce9391f8de0dd214c8a9943c
+size 4781670360

model-00019-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:054b725d32a33a51131aaf217436b0e0c1db02a0d7cf55802de139281ab886fe
+size 4964101416

model-00020-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:20ffea8d16473c3fa5398f2032b0e0e9f53d3382a7b081a285995b5f1c32e06a
+size 4781637360

model-00021-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e260ff1f77807109e7d48461f29ce10ace9446e6bc792c5ea9faa907d5c052ee
+size 4781670360

model-00022-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:50bbbe739cb8c17b373e9148c3b47ecf98771e960c2cf5a1829201156a3d2c88
+size 4781670360

model-00023-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4909c5c65ed4274a22b329c1980d437ab936f9ba217b6946b787cb39ec64c003
+size 4964101416

model-00024-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f7a06e7008fa871830090092a37b7f55163a78a1f906c5c1407af43682342abc
+size 4781637360

model-00025-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c9b5640b8bf11f1c6a3f50926695add6b4b3a393fd51fd2da8b10cf2a4e58ac8
+size 4781670360

model-00026-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:edeeee82b8f31dd865cf0dca92a074aaf9561c7424463b99431cd943fb575848
+size 4781670360

model-00027-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a6312e9994c2a33a834d3399debde12e329f08a8591ea03a0729bcb7a824c19f
+size 4964101416

model-00028-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f914a43f69e88e855f682fb746fc759ea93eea84c0c6ea5f36a702e6471c435f
+size 4781637360

model-00029-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b200b8f0d4bb7e98d783ba811fdd17c9179d456e5aff02752d9ac6739efcac42
+size 4781670360

model-00030-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:49ff5f411efb8357fb0f3d2944b552bdabfd204c547c3ea0cc95a50e8c68b49b
+size 4479675656

model-00031-of-00031.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:35f2930bd1f0d0156712681ba4c6517e456149f1815fb76a6652faf4fa87e565
+size 2491416704

model.safetensors.index.json ADDED Viewed

The diff for this file is too large to render. See raw diff

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,29 @@

+{
+  "do_convert_rgb": true,
+  "do_normalize": true,
+  "do_rescale": true,
+  "do_resize": true,
+  "image_mean": [
+    0.48145466,
+    0.4578275,
+    0.40821073
+  ],
+  "image_processor_type": "Qwen2VLImageProcessor",
+  "image_std": [
+    0.26862954,
+    0.26130258,
+    0.27577711
+  ],
+  "max_pixels": 2116800,
+  "merge_size": 2,
+  "min_pixels": 3136,
+  "patch_size": 14,
+  "processor_class": "Qwen2VLProcessor",
+  "resample": 3,
+  "rescale_factor": 0.00392156862745098,
+  "size": {
+    "max_pixels": 2116800,
+    "min_pixels": 3136
+  },
+  "temporal_patch_size": 2
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:091aa7594dc2fcfbfa06b9e3c22a5f0562ac14f30375c13af7309407a0e67b8a
+size 11420371

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,145 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "bos_token": null,
+  "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% for message in messages %}{{ message['role'].title() + ': ' + message['content'] | trim + eos_token + '\n' }}{% endfor %}{% if add_generation_prompt %}{{ 'Assistant: ' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "model_max_length": 32768,
+  "pad_token": "<|endoftext|>",
+  "padding_side": "right",
+  "processor_class": "Qwen2VLProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff