Visual Document Retrieval
Safetensors
sentence-transformers
colpali-engine
qwen3_5
vision-language
colbert
late-interaction
multi-vector
matryoshka
vidore
token-compression
Instructions to use tencent/EVIE-4.5B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sentence-transformers
How to use tencent/EVIE-4.5B with sentence-transformers:
from sentence_transformers import MultiVectorEncoder model = MultiVectorEncoder("tencent/EVIE-4.5B") queries = ["Which planet is known as the Red Planet?"] documents = [ "Venus is often called Earth's twin because of its similar size and proximity.", "Mars, known for its reddish appearance, is often referred to as the Red Planet.", "Jupiter, the largest planet in our solar system, has a prominent red spot.", ] query_embeddings = model.encode_query(queries) document_embeddings = model.encode_document(documents) similarities = model.similarity(query_embeddings, document_embeddings) print(similarities) - Notebooks
- Google Colab
- Kaggle
Integrate with Sentence Transformers via MultiVectorEncoder
Browse files- 1_Dense/config.json +8 -0
- 1_Dense/model.safetensors +3 -0
- 2_Normalize/config.json +4 -0
- 3_MultiVectorMask/config.json +3 -0
- README.md +53 -2
- additional_chat_templates/sentence_transformers.jinja +19 -0
- config_sentence_transformers.json +9 -0
- modules.json +26 -0
- processor_config.json +1 -1
- sentence_bert_config.json +30 -0
- tokenizer_config.json +2 -1
1_Dense/config.json
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"in_features": 2560,
|
| 3 |
+
"out_features": 2048,
|
| 4 |
+
"bias": true,
|
| 5 |
+
"activation_function": "torch.nn.modules.linear.Identity",
|
| 6 |
+
"module_input_name": "token_embeddings",
|
| 7 |
+
"module_output_name": "token_embeddings"
|
| 8 |
+
}
|
1_Dense/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0af96e01abcb912f81c4bd5cee1acb4460e0b97184b59f337be63cd15fc68631
|
| 3 |
+
size 20979880
|
2_Normalize/config.json
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"module_input_name": "token_embeddings",
|
| 3 |
+
"module_output_name": "token_embeddings"
|
| 4 |
+
}
|
3_MultiVectorMask/config.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"skiplist_words": []
|
| 3 |
+
}
|
README.md
CHANGED
|
@@ -13,6 +13,7 @@ tags:
|
|
| 13 |
- matryoshka
|
| 14 |
- vidore
|
| 15 |
- token-compression
|
|
|
|
| 16 |
datasets:
|
| 17 |
- vidore/vidore_benchmark
|
| 18 |
- vidore/vidore_benchmark_v2
|
|
@@ -191,7 +192,9 @@ Protocol `paired-all-pages-dedup+process_queries+ndcg2r-20260827` ($\text{MVT} =
|
|
| 191 |
|
| 192 |
## ⚡ Quick Start
|
| 193 |
|
| 194 |
-
###
|
|
|
|
|
|
|
| 195 |
|
| 196 |
```bash
|
| 197 |
git clone https://github.com/Tencent/EVIE.git
|
|
@@ -200,7 +203,7 @@ pip install -r requirements.txt
|
|
| 200 |
export PYTHONPATH="$(pwd)/colpali${PYTHONPATH:+:$PYTHONPATH}"
|
| 201 |
```
|
| 202 |
|
| 203 |
-
### Self-Contained Python Inference
|
| 204 |
|
| 205 |
```python
|
| 206 |
import torch
|
|
@@ -240,6 +243,54 @@ scores = processor.score(query_embeddings, image_embeddings)
|
|
| 240 |
print("Late-interaction MaxSim Relevance Score:", scores)
|
| 241 |
```
|
| 242 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 243 |
---
|
| 244 |
|
| 245 |
## 📂 Repository Layout
|
|
|
|
| 13 |
- matryoshka
|
| 14 |
- vidore
|
| 15 |
- token-compression
|
| 16 |
+
- sentence-transformers
|
| 17 |
datasets:
|
| 18 |
- vidore/vidore_benchmark
|
| 19 |
- vidore/vidore_benchmark_v2
|
|
|
|
| 192 |
|
| 193 |
## ⚡ Quick Start
|
| 194 |
|
| 195 |
+
### ColPali Engine
|
| 196 |
+
|
| 197 |
+
#### Installation
|
| 198 |
|
| 199 |
```bash
|
| 200 |
git clone https://github.com/Tencent/EVIE.git
|
|
|
|
| 203 |
export PYTHONPATH="$(pwd)/colpali${PYTHONPATH:+:$PYTHONPATH}"
|
| 204 |
```
|
| 205 |
|
| 206 |
+
#### Self-Contained Python Inference
|
| 207 |
|
| 208 |
```python
|
| 209 |
import torch
|
|
|
|
| 243 |
print("Late-interaction MaxSim Relevance Score:", scores)
|
| 244 |
```
|
| 245 |
|
| 246 |
+
### Sentence Transformers
|
| 247 |
+
|
| 248 |
+
Load EVIE with [Sentence Transformers](https://www.sbert.net/) to encode queries and document images and compute MaxSim scores. Bidirectional attention and the 768-token visual budget are configured automatically.
|
| 249 |
+
|
| 250 |
+
```bash
|
| 251 |
+
pip install -U "sentence-transformers[image]>=6.0.0"
|
| 252 |
+
```
|
| 253 |
+
|
| 254 |
+
```python
|
| 255 |
+
from sentence_transformers import MultiVectorEncoder
|
| 256 |
+
|
| 257 |
+
model = MultiVectorEncoder("tencent/EVIE-4.5B")
|
| 258 |
+
|
| 259 |
+
queries = [
|
| 260 |
+
"What is the variable represented on the y-axis of the graph?",
|
| 261 |
+
"Total outlay is maximum in which year?",
|
| 262 |
+
]
|
| 263 |
+
documents = [
|
| 264 |
+
"https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc1.jpg",
|
| 265 |
+
"https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc2.jpg",
|
| 266 |
+
"https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc3.jpg",
|
| 267 |
+
"https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc4.jpg",
|
| 268 |
+
]
|
| 269 |
+
|
| 270 |
+
query_embeddings = model.encode_query(queries)
|
| 271 |
+
document_embeddings = model.encode_document(documents)
|
| 272 |
+
print(query_embeddings[0].shape, document_embeddings[0].shape)
|
| 273 |
+
# torch.Size([23, 2048]) torch.Size([755, 2048])
|
| 274 |
+
|
| 275 |
+
scores = model.similarity(query_embeddings, document_embeddings)
|
| 276 |
+
print(scores)
|
| 277 |
+
# tensor([[17.0977, 7.8730, 6.9902, 4.8066],
|
| 278 |
+
# [ 4.4795, 12.8516, 4.2031, 4.3779]])
|
| 279 |
+
```
|
| 280 |
+
|
| 281 |
+
Documents can be URLs, local image paths, or `PIL.Image` objects. The example returns the full 2048-dimensional token embeddings. Scores can vary slightly with dtype and attention backend.
|
| 282 |
+
|
| 283 |
+
To use a smaller Matryoshka prefix, slice and renormalize each token embedding before scoring. Supported dimensions are 64, 128, 256, 512, 1024, and 2048:
|
| 284 |
+
|
| 285 |
+
```python
|
| 286 |
+
from torch.nn.functional import normalize
|
| 287 |
+
|
| 288 |
+
dimension = 128
|
| 289 |
+
query_embeddings = [normalize(embedding[..., :dimension], dim=-1) for embedding in query_embeddings]
|
| 290 |
+
document_embeddings = [normalize(embedding[..., :dimension], dim=-1) for embedding in document_embeddings]
|
| 291 |
+
scores = model.similarity(query_embeddings, document_embeddings)
|
| 292 |
+
```
|
| 293 |
+
|
| 294 |
---
|
| 295 |
|
| 296 |
## 📂 Repository Layout
|
additional_chat_templates/sentence_transformers.jinja
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- for message in messages -%}
|
| 2 |
+
{%- set ns = namespace(has_image=false, text='') -%}
|
| 3 |
+
{%- if message['content'] is string -%}
|
| 4 |
+
{%- set ns.text = message['content'] -%}
|
| 5 |
+
{%- else -%}
|
| 6 |
+
{%- for item in message['content'] -%}
|
| 7 |
+
{%- if 'image' in item or 'image_url' in item or item.type == 'image' -%}
|
| 8 |
+
{%- set ns.has_image = true -%}
|
| 9 |
+
{%- elif 'text' in item -%}
|
| 10 |
+
{%- set ns.text = ns.text + item.text -%}
|
| 11 |
+
{%- endif -%}
|
| 12 |
+
{%- endfor -%}
|
| 13 |
+
{%- endif -%}
|
| 14 |
+
{%- if ns.has_image -%}
|
| 15 |
+
{{- '<|im_start|>user\n<|vision_start|><|image_pad|><|vision_end|>Describe the image.<|im_end|><|endoftext|>' -}}
|
| 16 |
+
{%- else -%}
|
| 17 |
+
{{- ns.text + '<|endoftext|>' * 10 -}}
|
| 18 |
+
{%- endif -%}
|
| 19 |
+
{%- endfor -%}
|
config_sentence_transformers.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"__version__": {
|
| 3 |
+
"sentence_transformers": "6.0.0"
|
| 4 |
+
},
|
| 5 |
+
"model_type": "MultiVectorEncoder",
|
| 6 |
+
"similarity_fn_name": "maxsim",
|
| 7 |
+
"prompts": {},
|
| 8 |
+
"default_prompt_name": null
|
| 9 |
+
}
|
modules.json
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"idx": 0,
|
| 4 |
+
"name": "0",
|
| 5 |
+
"path": "",
|
| 6 |
+
"type": "sentence_transformers.base.modules.transformer.Transformer"
|
| 7 |
+
},
|
| 8 |
+
{
|
| 9 |
+
"idx": 1,
|
| 10 |
+
"name": "1",
|
| 11 |
+
"path": "1_Dense",
|
| 12 |
+
"type": "sentence_transformers.base.modules.dense.Dense"
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"idx": 2,
|
| 16 |
+
"name": "2",
|
| 17 |
+
"path": "2_Normalize",
|
| 18 |
+
"type": "sentence_transformers.sentence_transformer.modules.normalize.Normalize"
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"idx": 3,
|
| 22 |
+
"name": "3",
|
| 23 |
+
"path": "3_MultiVectorMask",
|
| 24 |
+
"type": "sentence_transformers.multi_vector_encoder.modules.multi_vector_mask.MultiVectorMask"
|
| 25 |
+
}
|
| 26 |
+
]
|
processor_config.json
CHANGED
|
@@ -25,7 +25,7 @@
|
|
| 25 |
},
|
| 26 |
"temporal_patch_size": 2
|
| 27 |
},
|
| 28 |
-
"processor_class": "
|
| 29 |
"video_processor": {
|
| 30 |
"do_convert_rgb": true,
|
| 31 |
"do_normalize": true,
|
|
|
|
| 25 |
},
|
| 26 |
"temporal_patch_size": 2
|
| 27 |
},
|
| 28 |
+
"processor_class": "Qwen3VLProcessor",
|
| 29 |
"video_processor": {
|
| 30 |
"do_convert_rgb": true,
|
| 31 |
"do_normalize": true,
|
sentence_bert_config.json
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"transformer_task": "feature-extraction",
|
| 3 |
+
"modality_config": {
|
| 4 |
+
"text": {
|
| 5 |
+
"method": "forward",
|
| 6 |
+
"method_output_name": "last_hidden_state"
|
| 7 |
+
},
|
| 8 |
+
"image": {
|
| 9 |
+
"method": "forward",
|
| 10 |
+
"method_output_name": "last_hidden_state"
|
| 11 |
+
},
|
| 12 |
+
"message": {
|
| 13 |
+
"method": "forward",
|
| 14 |
+
"method_output_name": "last_hidden_state",
|
| 15 |
+
"format": "structured"
|
| 16 |
+
}
|
| 17 |
+
},
|
| 18 |
+
"module_output_name": "token_embeddings",
|
| 19 |
+
"unpad_inputs": false,
|
| 20 |
+
"config_kwargs": {
|
| 21 |
+
"text_config": {
|
| 22 |
+
"is_causal": false
|
| 23 |
+
}
|
| 24 |
+
},
|
| 25 |
+
"processing_kwargs": {
|
| 26 |
+
"chat_template": {
|
| 27 |
+
"chat_template": "sentence_transformers"
|
| 28 |
+
}
|
| 29 |
+
}
|
| 30 |
+
}
|
tokenizer_config.json
CHANGED
|
@@ -24,7 +24,8 @@
|
|
| 24 |
},
|
| 25 |
"pad_token": "<|endoftext|>",
|
| 26 |
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 27 |
-
"processor_class": "
|
|
|
|
| 28 |
"split_special_tokens": false,
|
| 29 |
"tokenizer_class": "Qwen2Tokenizer",
|
| 30 |
"unk_token": null,
|
|
|
|
| 24 |
},
|
| 25 |
"pad_token": "<|endoftext|>",
|
| 26 |
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 27 |
+
"processor_class": "Qwen3VLProcessor",
|
| 28 |
+
"padding_side": "left",
|
| 29 |
"split_special_tokens": false,
|
| 30 |
"tokenizer_class": "Qwen2Tokenizer",
|
| 31 |
"unk_token": null,
|