Source: Phi35 Vision Demo · GitHub · microsoft/generative-ai-for-beginners Authors: Microsoft (microsoft) Licence: MIT — https://spdx.org/licenses/MIT.html
# pip install opencv-python
# import cv2
# import numpy as np
# def save_keyframes(video_path, output_folder):
# videoCapture = cv2.VideoCapture(video_path)
# success, frame = videoCapture.read()
# i = 0
# while success:
# gray_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
# hist = cv2.calcHist([gray_frame], [0], None, [256], [0, 256])
# success, next_frame = videoCapture.read()
# if not success:
# break
# next_gray_frame = cv2.cvtColor(next_frame, cv2.COLOR_BGR2GRAY)
# next_hist = cv2.calcHist([next_gray_frame], [0], None, [256], [0, 256])
# similarity = cv2.compareHist(hist, next_hist, cv2.HISTCMP_CORREL)
# if similarity < 0.9:
# i += 1
# cv2.imwrite(f"{output_folder}/keyframe_{i}.jpg", frame)
# print(f"Saved keyframe {i}")
# frame = next_frame
# videoCapture.release()
# save_keyframes('../video/copilot.mp4', '../output')
from PIL import Image
import requests, base64
images = []
placeholder = ""
for i in range(1,22):
with open("../output/keyframe_"+str(i)+".jpg", "rb") as f:
images.append(Image.open("../output/keyframe_"+str(i)+".jpg"))
placeholder += f"<|image_{i}|>\n"
# print(i)
images
from transformers import AutoModelForCausalLM
from transformers import AutoProcessor
model_id = "../Phi3Vision"
model = AutoModelForCausalLM.from_pretrained(model_id, device_map="cuda", trust_remote_code=True, torch_dtype="auto", _attn_implementation='flash_attention_2')
messages = [
{"role": "user", "content": placeholder+"Summarize the video."},
]
pip install transformers -U
from transformers import AutoModelForCausalLM
from transformers import AutoProcessor
# from image_embedding_phi3_v import Phi3VImageProcessor
# transformers.Phi3VImageProcessor = Phi3VImageProcessor
processor = AutoProcessor.from_pretrained(model_id, trust_remote_code=True, num_crops=4)
pip install jinja2 -U
prompt = processor.tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
inputs = processor(prompt, images, return_tensors="pt").to("cuda:0")
generation_args = { "max_new_tokens": 1000, "temperature": 0.0, "do_sample": False, }
generate_ids = model.generate(**inputs, eos_token_id=processor.tokenizer.eos_token_id, **generation_args)
generate_ids = generate_ids[:, inputs['input_ids'].shape[1]:]
response = processor.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
response
Disclaimer: This document has been translated using the AI translation service Co-op Translator. While we strive for accuracy, please be aware that automated translations may contain errors or inaccuracies. The original document in its native language should be considered the authoritative source. For critical information, professional human translation is recommended. We are not liable for any misunderstandings or misinterpretations arising from the use of this translation.