I only used the Gemini 1.5 Flash API and ran the following code. Although a single input doesn't exceed 400 tokens, the resulting invoice after a total of 700 requests doesn't match any of my calculations. Am I making a mistake somewhere? The code takes clips from the given video and sends them to the API to determine whether there is a woman or a man in the clip. According to my calculations:
700 requests * 400 tokens/request = 280,000 tokens 280,000 tokens / 1,000,000 tokens/$0.3 = 0.28 * $0.3 Total Cost = $0.084
However, according to the billing:
Generate Content input token count for Gemini 1.5 Flash when input is longer than 128k tokens - Total Usage - 11,993,645 count - Total Cost - 1.78$.
Generate Content input token count for Gemini 1.5 Flash when input is up to 128k tokens - Total Usage - 10,987,345 count - Total Cost - 0.82$.
class VideoGenderAnalyzer:
def __init__(self, api_key, video_path, interval=5, max_frames=100, debug_folder="debug_frames"):
self.api_key = api_key
self.video_path = video_path
self.interval = interval
self.max_frames = max_frames
self.debug_folder = debug_folder
# Configure the API
genai.configure(api_key=self.api_key)
# Ensure the debug folder exists
os.makedirs(self.debug_folder, exist_ok=True)
# Initialize counters
self.male_time = 0
self.female_time = 0
def extract_frames(self):
frames = []
cap = cv2.VideoCapture(self.video_path)
frame_count = 0
while frame_count < self.max_frames:
ret, frame = cap.read()
if not ret:
break
current_time = cap.get(cv2.CAP_PROP_POS_MSEC) / 1000
frames.append({
'frame': frame,
'timestamp': current_time
})
# Skip ahead
cap.set(cv2.CAP_PROP_POS_MSEC, (current_time + self.interval) * 1000)
frame_count += 1
cap.release()
return frames
def analyze_frame_gender(self, frame, index):
try:
# Removed the resizing step
# frame_resized = cv2.resize(frame, (1024, 1024), interpolation=cv2.INTER_AREA)
# Convert the image to base64
_, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 95])
image_base64 = base64.b64encode(buffer).decode('utf-8')
# Save the first 5 frames for debugging
if index < 5:
debug_path = os.path.join(self.debug_folder, f"frame_{index}.jpg")
cv2.imwrite(debug_path, frame, [cv2.IMWRITE_JPEG_QUALITY, 95])
print(f"Saved frame {index} to {debug_path}")
# Gemini API requires 'mime_type' and 'data'
input_data = {'mime_type': 'image/jpeg', 'data': image_base64}
model = genai.GenerativeModel(model_name="gemini-1.5-flash")
prompt = """In this image:
- How many men are there? (just the number)
- How many women are there? (just the number)
- If no one is on the screen, write 0
Provide the answer in this format:
Men: X
Women: Y"""
# API call
response = model.generate_content([input_data, prompt])
# Print the raw response for debugging
print(f"Response for frame {index}: {response.text}")
# Token bilgilerini yazdır
print(f"Frame {index} Token Bilgileri:")
print(f" Input Tokens: {response.usage_metadata.prompt_token_count}")
print(f" Output Tokens: {response.usage_metadata.candidates_token_count}")
print(f" Toplam Tokens: {response.usage_metadata.total_token_count}")
return response.text
except Exception as e:
print(f"Frame analysis error (frame {index}): {e}")
return "Men: 0\nWomen: 0"
The code's purpose is to send images to the API and have it return the number of women and men detected in each image. It does this, but it consumes/uses an excessive number of tokens.