def resize_image(image_path): img = Image.open(image_path) target_width, target_height = 640, 640 # Calculate the target size (maximum width and height). if target_width and target_height: max_size = (target_width, target_height) elif target_width: max_size = (target_width, img.height) elif target_height: max_size = (img.width, target_height) img.thumbnail(max_size) return img def get_model_response(img: Image, prompt: str, model, processor): # Prepare the messages for the model. messages = [ { "role": "system", "content": [{"type": "text", "text": "You are a helpful assistant. Reply only with the answer to the question asked in Luganda language only, and avoid using additional text in your response like 'here's the answer'."}] }, { "role": "user", "content": [ {"type": "image", "image": img}, {"type": "text", "text": prompt} ] } ] # Tokenize inputs and prepare for the model. inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt" ).to(model.device, dtype=torch.bfloat16) input_len = inputs["input_ids"].shape[-1] # Generate response from the model. with torch.inference_mode(): generation = model.generate(**inputs, max_new_tokens=100, do_sample=False) generation = generation[0][input_len:] # Decode the response. response = processor.decode(generation, skip_special_tokens=True) return response def extract_frames(video_path, num_frames): """ The function is adapted from: https://github.com/merveenoyan/smol-vision/blob/main/Gemma_3_for_Video_Understanding.ipynb """ cap = cv2.VideoCapture(video_path) if not cap.isOpened(): print("Error: Could not open video file.") return [] total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) fps = cap.get(cv2.CAP_PROP_FPS) # Calculate the step size to evenly distribute frames across the video. step = total_frames // num_frames frames = [] for i in range(num_frames): frame_idx = i * step cap.set(cv2.CAP_PROP_POS_FRAMES, frame_idx) ret, frame = cap.read() if not ret: break img = Image.fromarray(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)) timestamp = round(frame_idx / fps, 2) frames.append((img, timestamp)) cap.release() return frames def show_video(video_path, video_width = 600): video_file = open(video_path, "r+b").read() video_url = f"data:video/mp4;base64,{b64encode(video_file).decode()}" video_html = f"""""" return HTML(video_html) __ __