Videos + AI
Translated from the Spanish original. Read in Spanish
In this tutorial, you’ll learn how to take a video and turn it into a written tutorial using artificial intelligence. The following code shows how to extract frames from a video, convert them to base64 and use Azure OpenAI’s multimodal GPT-4o model to generate a tutorial based on those frames.

Prerequisites
Before you start, make sure you have the following in place:
- A
.envfile in the same directory as the script with the following information:
AZURE_TENANT_ID = "00000000-0000-0000-0000-000000000000"
AZURE_CLIENT_ID = "00000000-0000-0000-0000-000000000000"
AZURE_CLIENT_SECRET = "xxxxx"
- Install the required libraries:
opencv-python,python-dotenv,azure-identityandlangchain_openai.
Step 1: Set Up Credentials and Environment Variables
First, we load the credentials and set the environment variables needed to work with Azure OpenAI.
import base64
import os, cv2
from langchain_core.messages import HumanMessage
from langchain_openai import AzureChatOpenAI
from azure.identity import ChainedTokenCredential, EnvironmentCredential
from dotenv import load_dotenv
load_dotenv()
credential = ChainedTokenCredential(EnvironmentCredential())
access_token = credential.get_token("https://cognitiveservices.azure.com/.default")
deployment = "zerogap-openai-gpt4o"
embedding_deployment = "zerogap-openai-embedding"
os.environ["AZURE_OPENAI_ENDPOINT"] = "https://zero.openai.azure.com/"
os.environ["AZURE_OPENAI_API_KEY"] = access_token.token
os.environ["OPENAI_API_TYPE"] = "azure_ad"
os.environ["OPENAI_DEPLOYMENT"] = deployment
Step 2: Process the Video and Extract Frames
The process_video function opens the video, extracts frames at set intervals and converts them to base64.
def process_video(video_path, seconds_per_frame=2):
base64Frames = []
video = cv2.VideoCapture(video_path)
total_frames = int(video.get(cv2.CAP_PROP_FRAME_COUNT))
fps = video.get(cv2.CAP_PROP_FPS)
frames_to_skip = int(fps * seconds_per_frame)
curr_frame = 0
while curr_frame < total_frames - 1:
video.set(cv2.CAP_PROP_POS_FRAMES, curr_frame)
success, frame = video.read()
if not success:
break
_, buffer = cv2.imencode(".jpg", frame)
base64Frames.append(base64.b64encode(buffer).decode("utf-8"))
curr_frame += frames_to_skip
video.release()
print(f"Extracted {len(base64Frames)} frames")
return base64Frames
Step 3: Read the Video and Convert to Base64
In this step, we read a video from a file and process its frames. You can adjust the number of frames per second if the video is too long. The API accepts up to 20 images per request, so you’ll need to add some logic to process longer videos.
VIDEO_PATH = "C:\\demo.mp4"
base64Frames = process_video(VIDEO_PATH, seconds_per_frame=3)
Step 4: Prepare the Message for the Model
We build one text message, plus one for each frame of the video, to send to the model.
text_message = {
"type": "text",
"text": "Generate a tutorial based on a video. These are the frames from the video.",
}
video_message = [
*map(lambda x: {"type": "image_url",
"image_url": {"url": f'data:image/jpg;base64,{x}', "detail": "low"}}, base64Frames),
]
message_content = [text_message] + video_message
message = HumanMessage(content=message_content)
Step 5: Initialise the Model and Generate the Response
Finally, we initialise the Azure OpenAI model and generate the response.
llm = AzureChatOpenAI(openai_api_version="2024-02-01", azure_deployment=deployment, temperature=0.1)
output = llm.invoke([message])
print(output.content)
Complete Code
import base64
import os, cv2
from langchain_core.messages import HumanMessage
from langchain_openai import AzureChatOpenAI
from azure.identity import ChainedTokenCredential, EnvironmentCredential
from dotenv import load_dotenv
load_dotenv()
credential = ChainedTokenCredential(EnvironmentCredential())
access_token = credential.get_token("https://cognitiveservices.azure.com/.default")
deployment = "zerogap-openai-gpt4o"
embedding_deployment = "zerogap-openai-embedding"
os.environ["AZURE_OPENAI_ENDPOINT"] = "https://zero.openai.azure.com/"
os.environ["AZURE_OPENAI_API_KEY"] = access_token.token
os.environ["OPENAI_API_TYPE"] = "azure_ad"
os.environ["OPENAI_DEPLOYMENT"] = deployment
def process_video(video_path, seconds_per_frame=2):
base64Frames = []
video = cv2.VideoCapture(video_path)
total_frames = int(video.get(cv2.CAP_PROP_FRAME_COUNT))
fps = video.get(cv2.CAP_PROP_FPS)
frames_to_skip = int(fps * seconds_per_frame)
curr_frame = 0
while curr_frame < total_frames - 1:
video.set(cv2.CAP_PROP_POS_FRAMES, curr_frame)
success, frame = video.read()
if not success:
break
_, buffer = cv2.imencode(".jpg", frame)
base64Frames.append(base64.b64encode(buffer).decode("utf-8"))
curr_frame += frames_to_skip
video.release()
print(f"Extracted {len(base64Frames)} frames")
return base64Frames
VIDEO_PATH = "C:\\demo.mp4"
base64Frames = process_video(VIDEO_PATH, seconds_per_frame=3)
text_message = {
"type": "text",
"text": "Generate a tutorial based on a video. These are the frames from the video.",
}
video_message = [
*map(lambda x: {"type": "image_url",
"image_url": {"url": f'data:image/jpg;base64,{x}', "detail": "low"}}, base64Frames),
]
message_content = [text_message] + video_message
message = HumanMessage(content=message_content)
llm = AzureChatOpenAI(openai_api_version="2024-02-01", azure_deployment=deployment, temperature=0.1)
output = llm.invoke([message])
print(output.content)
