OER·harvester

← Back to the library
GitHub NOTEBOOK resource

Phi35 Vision Demo

Repository 21 Lessons, Get Started Building with Generative AI

Licence
OPEN MIT
Authors
Microsoft (microsoft)
Updated
2025-08-25 · GitHub
Language
en
Length
114 words
Open original ↗
{ }
Jupyter notebook Converted to a read-only view · code is not executed

Source: Phi35 Vision Demo · GitHub · microsoft/generative-ai-for-beginners Authors: Microsoft (microsoft) Licence: MIT — https://spdx.org/licenses/MIT.html

In [1]
# pip install opencv-python
In [2]
# import cv2
# import numpy as np
In [3]
# def save_keyframes(video_path, output_folder):
#     videoCapture = cv2.VideoCapture(video_path)
#     success, frame = videoCapture.read()
#     i = 0
#     while success:
#         gray_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
        
#         hist = cv2.calcHist([gray_frame], [0], None, [256], [0, 256])
        
#         success, next_frame = videoCapture.read()
#         if not success:
#             break
        
#         next_gray_frame = cv2.cvtColor(next_frame, cv2.COLOR_BGR2GRAY)
        
#         next_hist = cv2.calcHist([next_gray_frame], [0], None, [256], [0, 256])
        
#         similarity = cv2.compareHist(hist, next_hist, cv2.HISTCMP_CORREL)
        
#         if similarity < 0.9:
#             i += 1
#             cv2.imwrite(f"{output_folder}/keyframe_{i}.jpg", frame)
#             print(f"Saved keyframe {i}")
        
#         frame = next_frame

#     videoCapture.release()
In [4]
# save_keyframes('../video/copilot.mp4', '../output')
In [5]
from PIL import Image
import requests, base64
In [6]
images = [] 
placeholder = "" 
for i in range(1,22): 
    with open("../output/keyframe_"+str(i)+".jpg", "rb") as f:

        images.append(Image.open("../output/keyframe_"+str(i)+".jpg"))
        placeholder += f"<|image_{i}|>\n"
        # print(i)
In [7]
images
In [8]
from transformers import AutoModelForCausalLM 
from transformers import AutoProcessor
In [9]
model_id = "../Phi3Vision"
In [10]
model = AutoModelForCausalLM.from_pretrained(model_id, device_map="cuda", trust_remote_code=True, torch_dtype="auto", _attn_implementation='flash_attention_2')
In [11]
messages = [
                {"role": "user", "content": placeholder+"Summarize the video."}, 
]
In [12]
pip install transformers -U
In [13]
from transformers import AutoModelForCausalLM 
from transformers import AutoProcessor


# from image_embedding_phi3_v import Phi3VImageProcessor 

# transformers.Phi3VImageProcessor = Phi3VImageProcessor
In [14]
processor = AutoProcessor.from_pretrained(model_id, trust_remote_code=True, num_crops=4)
In [15]
pip install jinja2 -U
In [16]
prompt = processor.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
In [17]
inputs = processor(prompt, images, return_tensors="pt").to("cuda:0")
In [18]
generation_args = { "max_new_tokens": 1000, "temperature": 0.0, "do_sample": False, }
In [19]
generate_ids = model.generate(**inputs, eos_token_id=processor.tokenizer.eos_token_id, **generation_args)
In [20]
generate_ids = generate_ids[:, inputs['input_ids'].shape[1]:]
In [21]
response = processor.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
In [22]
response

Disclaimer: This document has been translated using the AI translation service Co-op Translator. While we strive for accuracy, please be aware that automated translations may contain errors or inaccuracies. The original document in its native language should be considered the authoritative source. For critical information, professional human translation is recommended. We are not liable for any misunderstandings or misinterpretations arising from the use of this translation.