QuentinJG tomaarsen HF Staff commited on
Commit
8131032
·
1 Parent(s): 810a3ed

Integrate with Sentence Transformers via MultiVectorEncoder (#5)

Browse files

- Integrate with Sentence Transformers via MultiVectorEncoder (9541b60607528dc5d1f475e72f7afd7b4b22ad12)


Co-authored-by: Tom Aarsen <tomaarsen@users.noreply.huggingface.co>

1_Dense/config.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "in_features": 768,
3
+ "out_features": 128,
4
+ "bias": true,
5
+ "activation_function": "torch.nn.modules.linear.Identity",
6
+ "module_input_name": "token_embeddings",
7
+ "module_output_name": "token_embeddings",
8
+ "use_residual": false
9
+ }
1_Dense/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e7ec5433ad5f7366368a1fdecb0245e9948f2b7ab4a60ceefe20413d8fddb6ce
3
+ size 393888
2_Normalize/config.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "module_input_name": "token_embeddings",
3
+ "module_output_name": "token_embeddings"
4
+ }
3_MultiVectorMask/config.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "skiplist_words": []
3
+ }
README.md CHANGED
@@ -8,6 +8,8 @@ tags:
8
  - colpali
9
  - vidore-experimental
10
  - vidore
 
 
11
  pipeline_tag: visual-document-retrieval
12
  ---
13
 
@@ -42,6 +44,48 @@ For more information about ModernVBERT, please check the [arXiv](https://arxiv.o
42
 
43
  ## Usage
44
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
45
  **🏎️ If your GPU supports it, we recommend using ModernVBERT with Flash Attention 2 to achieve the highest GPU throughput. To do so, install Flash Attention 2 as follows, then use the model as normal:**
46
 
47
  For now, the branch for using colmdernvbert is not yet merged in the official colpali repo, you need to clone the repo and checkout on the right branch to use it.
 
8
  - colpali
9
  - vidore-experimental
10
  - vidore
11
+ - sentence-transformers
12
+ - multi-vector
13
  pipeline_tag: visual-document-retrieval
14
  ---
15
 
 
44
 
45
  ## Usage
46
 
47
+ ### Sentence Transformers
48
+
49
+ ColModernVBERT can be loaded as a multi-vector (ColBERT-style late interaction) retriever with [Sentence Transformers](https://www.sbert.net/) via the `MultiVectorEncoder`, exposing the familiar `encode_query` / `encode_document` / `similarity` API.
50
+
51
+ ```bash
52
+ pip install "sentence-transformers[image]>=6.0.0"
53
+ ```
54
+
55
+ ```python
56
+ from sentence_transformers import MultiVectorEncoder
57
+
58
+ model = MultiVectorEncoder("ModernVBERT/colmodernvbert")
59
+
60
+ queries = [
61
+ "What is the variable represented on the y-axis of the graph?",
62
+ "Total outlay is maximum in which year?",
63
+ ]
64
+ images = [
65
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc1.jpg",
66
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc2.jpg",
67
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc3.jpg",
68
+ "https://huggingface.co/datasets/sentence-transformers/example-documents/resolve/main/doc4.jpg",
69
+ ]
70
+
71
+ query_embeddings = model.encode_query(queries, convert_to_tensor=True)
72
+ document_embeddings = model.encode_document(images, convert_to_tensor=True)
73
+ print(f"Query 0 shape: {tuple(query_embeddings[0].shape)}")
74
+ print(f"Document 0 shape: {tuple(document_embeddings[0].shape)}")
75
+ # Query 0 shape: (26, 128)
76
+ # Document 0 shape: (1149, 128)
77
+
78
+ scores = model.similarity(query_embeddings, document_embeddings)
79
+ print(scores)
80
+ # tensor([[16.7778, 10.3712, 11.8420, 9.0534],
81
+ # [ 7.3722, 12.0618, 8.1477, 7.9563]])
82
+ ```
83
+
84
+ > [!NOTE]
85
+ > `sentence_transformers.multi_vector_encoder.interpretability.get_n_patches` raises `NotImplementedError` for this model: like other Idefics3-style split-image processors, each page is split into sub-patch token blocks plus a global patch, so the token grid is not a simple rectangle.
86
+
87
+ ### ColPali Engine
88
+
89
  **🏎️ If your GPU supports it, we recommend using ModernVBERT with Flash Attention 2 to achieve the highest GPU throughput. To do so, install Flash Attention 2 as follows, then use the model as normal:**
90
 
91
  For now, the branch for using colmdernvbert is not yet merged in the official colpali repo, you need to clone the repo and checkout on the right branch to use it.
adapter_config.json CHANGED
@@ -24,7 +24,7 @@
24
  "r": 32,
25
  "rank_pattern": {},
26
  "revision": null,
27
- "target_modules": "(.*(model.text_model).*(Wo|Wqkv|Wi).*$|.*(custom_text_proj).*$)",
28
  "task_type": "FEATURE_EXTRACTION",
29
  "trainable_token_indices": null,
30
  "use_dora": false,
 
24
  "r": 32,
25
  "rank_pattern": {},
26
  "revision": null,
27
+ "target_modules": "(.*(text_model).*(Wo|Wqkv|Wi).*$|.*(custom_text_proj).*$)",
28
  "task_type": "FEATURE_EXTRACTION",
29
  "trainable_token_indices": null,
30
  "use_dora": false,
additional_chat_templates/sentence_transformers.jinja ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- for message in messages -%}
2
+ {%- if message['content'] is string -%}
3
+ {%- if task is defined and task == 'query' -%}
4
+ {{- message['content'] -}}
5
+ {%- for _ in range(10) -%}{{- '<end_of_utterance>' -}}{%- endfor -%}
6
+ {%- else -%}
7
+ {{- message['content'] -}}
8
+ {%- endif -%}
9
+ {%- else -%}
10
+ {%- for content in message['content'] -%}
11
+ {%- if content['type'] == 'image' -%}
12
+ {{- '<|begin_of_text|>User:<image>Describe the image.<end_of_utterance>\nAssistant:' -}}
13
+ {%- elif content['type'] == 'text' -%}
14
+ {%- if task is defined and task == 'query' -%}
15
+ {{- content['text'] -}}
16
+ {%- for _ in range(10) -%}{{- '<end_of_utterance>' -}}{%- endfor -%}
17
+ {%- else -%}
18
+ {{- content['text'] -}}
19
+ {%- endif -%}
20
+ {%- endif -%}
21
+ {%- endfor -%}
22
+ {%- endif -%}
23
+ {%- endfor -%}
config_sentence_transformers.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "__version__": {
3
+ "sentence_transformers": "6.0.0"
4
+ },
5
+ "default_prompt_name": null,
6
+ "model_type": "MultiVectorEncoder",
7
+ "requirements": {
8
+ "transformers": {
9
+ "specifier": ">=5.15",
10
+ "reason": "Older versions ignore the key_mapping, which silently randomizes the adapter weights."
11
+ }
12
+ },
13
+ "prompts": {
14
+ "document": "",
15
+ "query": ""
16
+ },
17
+ "similarity_fn_name": null
18
+ }
modules.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "idx": 0,
4
+ "name": "0",
5
+ "path": "",
6
+ "type": "sentence_transformers.base.modules.transformer.Transformer"
7
+ },
8
+ {
9
+ "idx": 1,
10
+ "name": "1",
11
+ "path": "1_Dense",
12
+ "type": "sentence_transformers.base.modules.dense.Dense"
13
+ },
14
+ {
15
+ "idx": 2,
16
+ "name": "2",
17
+ "path": "2_Normalize",
18
+ "type": "sentence_transformers.sentence_transformer.modules.normalize.Normalize"
19
+ },
20
+ {
21
+ "idx": 3,
22
+ "name": "3",
23
+ "path": "3_MultiVectorMask",
24
+ "type": "sentence_transformers.multi_vector_encoder.modules.multi_vector_mask.MultiVectorMask"
25
+ }
26
+ ]
preprocessor_config.json CHANGED
@@ -19,7 +19,7 @@
19
  "max_image_size": {
20
  "longest_edge": 512
21
  },
22
- "processor_class": "ColModernVBertProcessor",
23
  "resample": 1,
24
  "rescale_factor": 0.00392156862745098,
25
  "size": {
 
19
  "max_image_size": {
20
  "longest_edge": 512
21
  },
22
+ "processor_class": "Idefics3Processor",
23
  "resample": 1,
24
  "rescale_factor": 0.00392156862745098,
25
  "size": {
processor_config.json CHANGED
@@ -1,4 +1,4 @@
1
  {
2
  "image_seq_len": 64,
3
- "processor_class": "ColModernVBertProcessor"
4
  }
 
1
  {
2
  "image_seq_len": 64,
3
+ "processor_class": "Idefics3Processor"
4
  }
sentence_bert_config.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "transformer_task": "feature-extraction",
3
+ "modality_config": {
4
+ "text": {
5
+ "method": "forward",
6
+ "method_output_name": "last_hidden_state"
7
+ },
8
+ "image": {
9
+ "method": "forward",
10
+ "method_output_name": "last_hidden_state"
11
+ },
12
+ "message": {
13
+ "method": "forward",
14
+ "method_output_name": "last_hidden_state",
15
+ "format": "structured"
16
+ }
17
+ },
18
+ "module_output_name": "token_embeddings",
19
+ "unpad_inputs": false,
20
+ "model_kwargs": {
21
+ "key_mapping": {
22
+ "^model\\.": ""
23
+ }
24
+ },
25
+ "processing_kwargs": {
26
+ "chat_template": {
27
+ "chat_template": "sentence_transformers"
28
+ }
29
+ }
30
+ }
tokenizer_config.json CHANGED
@@ -1271,7 +1271,7 @@
1271
  "pad_token": "[PAD]",
1272
  "pad_token_type_id": 0,
1273
  "padding_side": "left",
1274
- "processor_class": "ColModernVBertProcessor",
1275
  "sep_token": "[SEP]",
1276
  "stride": 0,
1277
  "tokenizer_class": "PreTrainedTokenizerFast",
 
1271
  "pad_token": "[PAD]",
1272
  "pad_token_type_id": 0,
1273
  "padding_side": "left",
1274
+ "processor_class": "Idefics3Processor",
1275
  "sep_token": "[SEP]",
1276
  "stride": 0,
1277
  "tokenizer_class": "PreTrainedTokenizerFast",