RhapsodyAI
/

MiniCPM-V-Embedding-preview

@@ -56,8 +56,8 @@ import torch
 device = 'cuda:0'
-def last_token_pool(last_hidden_states: Tensor,
-                 attention_mask: Tensor) -> Tensor:
     left_padding = (attention_mask[:, -1].sum() == attention_mask.shape[0])
     if left_padding:
         return last_hidden_states[:, -1]
@@ -66,30 +66,36 @@ def last_token_pool(last_hidden_states: Tensor,
         batch_size = last_hidden_states.shape[0]
         return last_hidden_states[torch.arange(batch_size, device=last_hidden_states.device), sequence_lengths]
-tokenizer = AutoTokenizer.from_pretrained('/local/path/to/minicpm-visual-embedding-v0', trust_remote_code=True)
-model = AutoModel.from_pretrained('/local/path/to/minicpm-visual-embedding-v0', trust_remote_code=True)
-image_1 = Image.open('/local/path/to/document1.png').convert('RGB')
-image_2 = Image.open('/local/path/to/document2.png').convert('RGB')
 query_instruction = 'Represent this query for retrieving relavant document: '
 query = 'Who was elected as president of United States in 2020?'
 query_full = query_instruction + query
-# Embed text queries
-q_outputs = model(text=[query_full], image=[None, None], tokenizer=tokenizer) # [B, s, d]
-q_reps = last_token_pool(q_outputs.last_hidden_state, q_outputs.attention_mask) # [B, d]
 # Embed image documents
-p_outputs = model(text=['', ''], image=[image_1, image_2], tokenizer=tokenizer) # [B, s, d]
-p_reps = last_token_pool(p_outputs.last_hidden_state, p_outputs.attention_mask) # [B, d]
-# Calculate similarities
-scores = torch.matmul(q_reps, p_reps)
 print(scores)
 ```

 device = 'cuda:0'
+# This function is borrowed from https://huggingface.co/intfloat/e5-mistral-7b-instruct
+def last_token_pool(last_hidden_states, attention_mask):
     left_padding = (attention_mask[:, -1].sum() == attention_mask.shape[0])
     if left_padding:
         return last_hidden_states[:, -1]
         batch_size = last_hidden_states.shape[0]
         return last_hidden_states[torch.arange(batch_size, device=last_hidden_states.device), sequence_lengths]
+# Load model, be sure to substitute `model_path` by your model path
+model_path = '/local/path/to/model'
+tokenizer = AutoTokenizer.from_pretrained(model_path, trust_remote_code=True)
+model = AutoModel.from_pretrained(model_path, trust_remote_code=True)
+model.to(device)
+# Load image to PIL.Image object
+image_1 = Image.open('/local/path/to/images/memex.png').convert('RGB')
+image_2 = Image.open('/local/path/to/images/us2020.png').convert('RGB')
+image_3 = Image.open('/local/path/to/images/hard_negative.png').convert('RGB')
+# User query
 query_instruction = 'Represent this query for retrieving relavant document: '
 query = 'Who was elected as president of United States in 2020?'
 query_full = query_instruction + query
 # Embed image documents
+with torch.no_grad():
+    p_outputs = model(text=['', '', ''], image=[image_1, image_2, image_3], tokenizer=tokenizer)
+    p_reps = last_token_pool(p_outputs.last_hidden_state, p_outputs.attention_mask)
+# Embed text queries
+with torch.no_grad():
+    q_outputs = model(text=[query_full], image=[None], tokenizer=tokenizer) # [B, s, d]
+    q_reps = last_token_pool(q_outputs.last_hidden_state, q_outputs.attention_mask) # [B, d]
+# Calculate similarities
+scores = torch.matmul(q_reps, p_reps.T)
 print(scores)
+# tensor([[0.6506, 4.9630, 3.8614]], device='cuda:0')
 ```