Gemini's excessively high token consumption issue
13:11 30 Nov 2024

I only used the Gemini 1.5 Flash API and ran the following code. Although a single input doesn't exceed 400 tokens, the resulting invoice after a total of 700 requests doesn't match any of my calculations. Am I making a mistake somewhere? The code takes clips from the given video and sends them to the API to determine whether there is a woman or a man in the clip. According to my calculations:

700 requests * 400 tokens/request = 280,000 tokens 280,000 tokens / 1,000,000 tokens/$0.3 = 0.28 * $0.3 Total Cost = $0.084

However, according to the billing:

Generate Content input token count for Gemini 1.5 Flash when input is longer than 128k tokens - Total Usage - 11,993,645 count - Total Cost - 1.78$.

Generate Content input token count for Gemini 1.5 Flash when input is up to 128k tokens - Total Usage - 10,987,345 count - Total Cost - 0.82$.

class VideoGenderAnalyzer:
        def __init__(self, api_key, video_path, interval=5, max_frames=100, debug_folder="debug_frames"):

            self.api_key = api_key
            self.video_path = video_path
            self.interval = interval
            self.max_frames = max_frames
            self.debug_folder = debug_folder

            # Configure the API
            genai.configure(api_key=self.api_key)

            # Ensure the debug folder exists
            os.makedirs(self.debug_folder, exist_ok=True)

            # Initialize counters
            self.male_time = 0
            self.female_time = 0

        def extract_frames(self):

            frames = []
            cap = cv2.VideoCapture(self.video_path)

            frame_count = 0
            while frame_count < self.max_frames:
                ret, frame = cap.read()
                if not ret:
                    break

                current_time = cap.get(cv2.CAP_PROP_POS_MSEC) / 1000
                frames.append({
                    'frame': frame,
                    'timestamp': current_time
                })

                # Skip ahead
                cap.set(cv2.CAP_PROP_POS_MSEC, (current_time + self.interval) * 1000)
                frame_count += 1

            cap.release()
            return frames

        def analyze_frame_gender(self, frame, index):

            try:
                # Removed the resizing step
                # frame_resized = cv2.resize(frame, (1024, 1024), interpolation=cv2.INTER_AREA)

                # Convert the image to base64
                _, buffer = cv2.imencode('.jpg', frame, [cv2.IMWRITE_JPEG_QUALITY, 95])
                image_base64 = base64.b64encode(buffer).decode('utf-8')

                # Save the first 5 frames for debugging
                if index < 5:
                    debug_path = os.path.join(self.debug_folder, f"frame_{index}.jpg")
                    cv2.imwrite(debug_path, frame, [cv2.IMWRITE_JPEG_QUALITY, 95])
                    print(f"Saved frame {index} to {debug_path}")

                # Gemini API requires 'mime_type' and 'data'
                input_data = {'mime_type': 'image/jpeg', 'data': image_base64}

                model = genai.GenerativeModel(model_name="gemini-1.5-flash")

                prompt = """In this image:
        - How many men are there? (just the number)
        - How many women are there? (just the number)
        - If no one is on the screen, write 0

        Provide the answer in this format:
        Men: X
        Women: Y"""

                # API call
                response = model.generate_content([input_data, prompt])

                # Print the raw response for debugging
                print(f"Response for frame {index}: {response.text}")
                # Token bilgilerini yazdır
                print(f"Frame {index} Token Bilgileri:")
                print(f"  Input Tokens: {response.usage_metadata.prompt_token_count}")
                print(f"  Output Tokens: {response.usage_metadata.candidates_token_count}")
                print(f"  Toplam Tokens: {response.usage_metadata.total_token_count}")
                return response.text
            except Exception as e:
                print(f"Frame analysis error (frame {index}): {e}")
                return "Men: 0\nWomen: 0"

The code's purpose is to send images to the API and have it return the number of women and men detected in each image. It does this, but it consumes/uses an excessive number of tokens.

image large-language-model google-gemini google-generativeai