OER·harvester

← Back to the library
GitHub NOTEBOOK resource

Phi35 Vision Demo

Repository 21 Lessons, Get Started Building with Generative AI

Licence
OPEN MIT
Authors
Microsoft (microsoft)
Updated
2024-09-12 · GitHub
Language
en detected
Length
38 words
Open original ↗
{ }
Jupyter notebook Converted to a read-only view · code is not executed

Source: Phi35 Vision Demo · GitHub · microsoft/generative-ai-for-beginners Authors: Microsoft (microsoft) Licence: MIT — https://spdx.org/licenses/MIT.html

In [1]
# pip install opencv-python
In [2]
# import cv2
# import numpy as np
In [3]
# def save_keyframes(video_path, output_folder):
#     videoCapture = cv2.VideoCapture(video_path)
#     success, frame = videoCapture.read()
#     i = 0
#     while success:
#         gray_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
        
#         hist = cv2.calcHist([gray_frame], [0], None, [256], [0, 256])
        
#         success, next_frame = videoCapture.read()
#         if not success:
#             break
        
#         next_gray_frame = cv2.cvtColor(next_frame, cv2.COLOR_BGR2GRAY)
        
#         next_hist = cv2.calcHist([next_gray_frame], [0], None, [256], [0, 256])
        
#         similarity = cv2.compareHist(hist, next_hist, cv2.HISTCMP_CORREL)
        
#         if similarity < 0.9:
#             i += 1
#             cv2.imwrite(f"{output_folder}/keyframe_{i}.jpg", frame)
#             print(f"Saved keyframe {i}")
        
#         frame = next_frame

#     videoCapture.release()
In [4]
# save_keyframes('../video/copilot.mp4', '../output')
In [5]
from PIL import Image
import requests, base64
In [6]
images = [] 
placeholder = "" 
for i in range(1,22): 
    with open("../output/keyframe_"+str(i)+".jpg", "rb") as f:

        images.append(Image.open("../output/keyframe_"+str(i)+".jpg"))
        placeholder += f"<|image_{i}|>\n"
        # print(i)
In [7]
images
In [8]
from transformers import AutoModelForCausalLM 
from transformers import AutoProcessor
In [9]
model_id = "../Phi3Vision"
In [10]
model = AutoModelForCausalLM.from_pretrained(model_id, device_map="cuda", trust_remote_code=True, torch_dtype="auto", _attn_implementation='flash_attention_2')
In [11]
messages = [
                {"role": "user", "content": placeholder+"Summarize the video."}, 
]
In [12]
pip install transformers -U
In [13]
from transformers import AutoModelForCausalLM 
from transformers import AutoProcessor


# from image_embedding_phi3_v import Phi3VImageProcessor 

# transformers.Phi3VImageProcessor = Phi3VImageProcessor
In [14]
processor = AutoProcessor.from_pretrained(model_id, trust_remote_code=True, num_crops=4)
In [15]
pip install jinja2 -U
In [16]
prompt = processor.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
In [17]
inputs = processor(prompt, images, return_tensors="pt").to("cuda:0")
In [18]
generation_args = { "max_new_tokens": 1000, "temperature": 0.0, "do_sample": False, }
In [19]
generate_ids = model.generate(**inputs, eos_token_id=processor.tokenizer.eos_token_id, **generation_args)
In [20]
generate_ids = generate_ids[:, inputs['input_ids'].shape[1]:]
In [21]
response = processor.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
In [22]
response