tomaarsen HF Staff commited on
Commit
3fe2eb5
·
verified ·
1 Parent(s): 355deaf

Integrate with Sentence Transformers via MultiVectorEncoder

Browse files
1_Dense/config.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "in_features": 2560,
3
+ "out_features": 2048,
4
+ "bias": true,
5
+ "activation_function": "torch.nn.modules.linear.Identity",
6
+ "module_input_name": "token_embeddings",
7
+ "module_output_name": "token_embeddings"
8
+ }
1_Dense/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0af96e01abcb912f81c4bd5cee1acb4460e0b97184b59f337be63cd15fc68631
3
+ size 20979880
2_Normalize/config.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "module_input_name": "token_embeddings",
3
+ "module_output_name": "token_embeddings"
4
+ }
3_MultiVectorMask/config.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "skiplist_words": []
3
+ }
README.md CHANGED
@@ -13,6 +13,7 @@ tags:
13
  - matryoshka
14
  - vidore
15
  - token-compression
 
16
  datasets:
17
  - vidore/vidore_benchmark
18
  - vidore/vidore_benchmark_v2
@@ -191,7 +192,9 @@ Protocol `paired-all-pages-dedup+process_queries+ndcg2r-20260827` ($\text{MVT} =
191
 
192
  ## ⚡ Quick Start
193
 
194
- ### Installation
 
 
195
 
196
  ```bash
197
  git clone https://github.com/Tencent/EVIE.git
@@ -200,7 +203,7 @@ pip install -r requirements.txt
200
  export PYTHONPATH="$(pwd)/colpali${PYTHONPATH:+:$PYTHONPATH}"
201
  ```
202
 
203
- ### Self-Contained Python Inference
204
 
205
  ```python
206
  import torch
@@ -240,6 +243,54 @@ scores = processor.score(query_embeddings, image_embeddings)
240
  print("Late-interaction MaxSim Relevance Score:", scores)
241
  ```
242
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
243
  ---
244
 
245
  ## 📂 Repository Layout
 
13
  - matryoshka
14
  - vidore
15
  - token-compression
16
+ - sentence-transformers
17
  datasets:
18
  - vidore/vidore_benchmark
19
  - vidore/vidore_benchmark_v2
 
192
 
193
  ## ⚡ Quick Start
194
 
195
+ ### ColPali Engine
196
+
197
+ #### Installation
198
 
199
  ```bash
200
  git clone https://github.com/Tencent/EVIE.git
 
203
  export PYTHONPATH="$(pwd)/colpali${PYTHONPATH:+:$PYTHONPATH}"
204
  ```
205
 
206
+ #### Self-Contained Python Inference
207
 
208
  ```python
209
  import torch
 
243
  print("Late-interaction MaxSim Relevance Score:", scores)
244
  ```
245
 
246
+ ### Sentence Transformers
247
+
248
+ Load EVIE with [Sentence Transformers](https://www.sbert.net/) to encode queries and document images and compute MaxSim scores. Bidirectional attention and the 768-token visual budget are configured automatically.
249
+
250
+ ```bash
251
+ pip install -U "sentence-transformers[image]>=6.0.0"
252
+ ```
253
+
254
+ ```python
255
+ from sentence_transformers import MultiVectorEncoder
256
+
257
+ model = MultiVectorEncoder("tencent/EVIE-4.5B")
258
+
259
+ queries = [
260
+ "What is the variable represented on the y-axis of the graph?",
261
+ "Total outlay is maximum in which year?",
262
+ ]
263
+ documents = [
264
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc1.jpg",
265
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc2.jpg",
266
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc3.jpg",
267
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc4.jpg",
268
+ ]
269
+
270
+ query_embeddings = model.encode_query(queries)
271
+ document_embeddings = model.encode_document(documents)
272
+ print(query_embeddings[0].shape, document_embeddings[0].shape)
273
+ # torch.Size([23, 2048]) torch.Size([755, 2048])
274
+
275
+ scores = model.similarity(query_embeddings, document_embeddings)
276
+ print(scores)
277
+ # tensor([[17.0977, 7.8730, 6.9902, 4.8066],
278
+ # [ 4.4795, 12.8516, 4.2031, 4.3779]])
279
+ ```
280
+
281
+ Documents can be URLs, local image paths, or `PIL.Image` objects. The example returns the full 2048-dimensional token embeddings. Scores can vary slightly with dtype and attention backend.
282
+
283
+ To use a smaller Matryoshka prefix, slice and renormalize each token embedding before scoring. Supported dimensions are 64, 128, 256, 512, 1024, and 2048:
284
+
285
+ ```python
286
+ from torch.nn.functional import normalize
287
+
288
+ dimension = 128
289
+ query_embeddings = [normalize(embedding[..., :dimension], dim=-1) for embedding in query_embeddings]
290
+ document_embeddings = [normalize(embedding[..., :dimension], dim=-1) for embedding in document_embeddings]
291
+ scores = model.similarity(query_embeddings, document_embeddings)
292
+ ```
293
+
294
  ---
295
 
296
  ## 📂 Repository Layout
additional_chat_templates/sentence_transformers.jinja ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- for message in messages -%}
2
+ {%- set ns = namespace(has_image=false, text='') -%}
3
+ {%- if message['content'] is string -%}
4
+ {%- set ns.text = message['content'] -%}
5
+ {%- else -%}
6
+ {%- for item in message['content'] -%}
7
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' -%}
8
+ {%- set ns.has_image = true -%}
9
+ {%- elif 'text' in item -%}
10
+ {%- set ns.text = ns.text + item.text -%}
11
+ {%- endif -%}
12
+ {%- endfor -%}
13
+ {%- endif -%}
14
+ {%- if ns.has_image -%}
15
+ {{- '<|im_start|>user\n<|vision_start|><|image_pad|><|vision_end|>Describe the image.<|im_end|><|endoftext|>' -}}
16
+ {%- else -%}
17
+ {{- ns.text + '<|endoftext|>' * 10 -}}
18
+ {%- endif -%}
19
+ {%- endfor -%}
config_sentence_transformers.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "__version__": {
3
+ "sentence_transformers": "6.0.0"
4
+ },
5
+ "model_type": "MultiVectorEncoder",
6
+ "similarity_fn_name": "maxsim",
7
+ "prompts": {},
8
+ "default_prompt_name": null
9
+ }
modules.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "idx": 0,
4
+ "name": "0",
5
+ "path": "",
6
+ "type": "sentence_transformers.base.modules.transformer.Transformer"
7
+ },
8
+ {
9
+ "idx": 1,
10
+ "name": "1",
11
+ "path": "1_Dense",
12
+ "type": "sentence_transformers.base.modules.dense.Dense"
13
+ },
14
+ {
15
+ "idx": 2,
16
+ "name": "2",
17
+ "path": "2_Normalize",
18
+ "type": "sentence_transformers.sentence_transformer.modules.normalize.Normalize"
19
+ },
20
+ {
21
+ "idx": 3,
22
+ "name": "3",
23
+ "path": "3_MultiVectorMask",
24
+ "type": "sentence_transformers.multi_vector_encoder.modules.multi_vector_mask.MultiVectorMask"
25
+ }
26
+ ]
processor_config.json CHANGED
@@ -25,7 +25,7 @@
25
  },
26
  "temporal_patch_size": 2
27
  },
28
- "processor_class": "ColQwen3_5Processor",
29
  "video_processor": {
30
  "do_convert_rgb": true,
31
  "do_normalize": true,
 
25
  },
26
  "temporal_patch_size": 2
27
  },
28
+ "processor_class": "Qwen3VLProcessor",
29
  "video_processor": {
30
  "do_convert_rgb": true,
31
  "do_normalize": true,
sentence_bert_config.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "transformer_task": "feature-extraction",
3
+ "modality_config": {
4
+ "text": {
5
+ "method": "forward",
6
+ "method_output_name": "last_hidden_state"
7
+ },
8
+ "image": {
9
+ "method": "forward",
10
+ "method_output_name": "last_hidden_state"
11
+ },
12
+ "message": {
13
+ "method": "forward",
14
+ "method_output_name": "last_hidden_state",
15
+ "format": "structured"
16
+ }
17
+ },
18
+ "module_output_name": "token_embeddings",
19
+ "unpad_inputs": false,
20
+ "config_kwargs": {
21
+ "text_config": {
22
+ "is_causal": false
23
+ }
24
+ },
25
+ "processing_kwargs": {
26
+ "chat_template": {
27
+ "chat_template": "sentence_transformers"
28
+ }
29
+ }
30
+ }
tokenizer_config.json CHANGED
@@ -24,7 +24,8 @@
24
  },
25
  "pad_token": "<|endoftext|>",
26
  "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
27
- "processor_class": "ColQwen3_5Processor",
 
28
  "split_special_tokens": false,
29
  "tokenizer_class": "Qwen2Tokenizer",
30
  "unk_token": null,
 
24
  },
25
  "pad_token": "<|endoftext|>",
26
  "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
27
+ "processor_class": "Qwen3VLProcessor",
28
+ "padding_side": "left",
29
  "split_special_tokens": false,
30
  "tokenizer_class": "Qwen2Tokenizer",
31
  "unk_token": null,