File size: 3,585 Bytes
a31f556
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
# backend/tools/captcha_solver.py
import logging
import asyncio
import json

from backend.services.token_manager import gemini_call_with_checkpoint

async def solve_captcha_if_present(page):
    """
    Scans the DOM for reCAPTCHA, hCaptcha, or Cloudflare Turnstile.
    If found, takes a screenshot and uses gemini-3.5-flash to locate the exact click coordinates.
    """
    try:
        # Quick check: does the page contain common CAPTCHA keywords in frames?
        frames = page.frames
        captcha_detected = False
        for frame in frames:
            name = frame.name.lower()
            url = frame.url.lower()
            if "recaptcha" in name or "recaptcha" in url or "hcaptcha" in url or "turnstile" in url:
                captcha_detected = True
                break
                
        if not captcha_detected:
            # Also check page content for challenge iframes
            content = await page.content()
            if "recaptcha" in content.lower() or "hcaptcha" in content.lower() or "turnstile" in content.lower():
                captcha_detected = True

        if not captcha_detected:
            return False

        logging.warning("CAPTCHA detected! Engaging Gemini 3.5 Flash Vision solver...")
        
        # Max 3 attempts for dynamic puzzles
        for attempt in range(3):
            # Take full page screenshot
            screenshot_bytes = await page.screenshot()
            
            prompt = """You are an advanced CAPTCHA solver. Look at this screenshot. 
Find the CAPTCHA element (like an 'I am human' checkbox, or a puzzle tile that needs to be clicked).
Return ONLY a valid JSON object containing the exact absolute X and Y pixel coordinates representing the center of the element to click.
Example: {"x": 450, "y": 600}
If the CAPTCHA is already solved or no CAPTCHA is visible, return {"x": 0, "y": 0}.
Do not include markdown blocks, just the raw JSON.
"""
            # Encode screenshot as base64 for the vision prompt
            import base64
            img_b64 = base64.b64encode(screenshot_bytes).decode()
            vision_prompt = f"{prompt}\n\n[IMAGE_BASE64]: data:image/png;base64,{img_b64[:500]}..." 
            
            try:
                raw_text = await gemini_call_with_checkpoint(
                    prompt=vision_prompt,
                    task_type="captcha",
                    persona="jarvis"
                )
                raw_text = raw_text.strip().strip("```json").strip("```").strip()
                coords = json.loads(raw_text)
                x = int(coords.get('x', 0))
                y = int(coords.get('y', 0))
                
                if x == 0 and y == 0:
                    logging.info("Gemini 3.5 Flash reported CAPTCHA cleared or not visible.")
                    return True
                    
                logging.info(f"Gemini 3.5 Flash requested click at ({x}, {y}). Executing stealth move...")
                # Stealth mouse movement
                await page.mouse.move(x, y, steps=10)
                await asyncio.sleep(0.5)
                await page.mouse.click(x, y)
                
                # Wait for puzzle transition or success mark
                await asyncio.sleep(4)
                
            except Exception as genai_err:
                logging.error(f"Gemini Vision parsing failed on attempt {attempt+1}: {genai_err}")
                await asyncio.sleep(2)
                
        return True
    except Exception as e:
        logging.error(f"CAPTCHA solver master error: {e}")
        return False