#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
QuantumToken 本地 AI 自动化 Agent
在本地电脑运行，接收任务后自动截图、调用多模态模型决策、执行鼠标键盘操作。

依赖：
  pip install pyautogui Pillow requests

用法：
  python agent.py --api-key "你的QuantumToken API Key" --task "打开Chrome搜索QuantumToken"
"""

import argparse
import base64
import io
import json
import os
import re
import sys
import time
import traceback
from typing import Optional

import pyautogui
import requests
from PIL import Image

# 安全设置：鼠标移到屏幕角落不触发 fail-safe 暂停
pyautogui.FAILSAFE = True


def capture_screenshot() -> str:
    """截取屏幕并返回 base64 PNG"""
    screenshot = pyautogui.screenshot()
    buffer = io.BytesIO()
    screenshot.save(buffer, format="PNG")
    return base64.b64encode(buffer.getvalue()).decode("utf-8")


def build_messages(task: str, history: list, image_b64: str) -> list:
    """构建发送给多模态模型的消息"""
    system_prompt = """你是一个电脑自动化操作助手。你会收到当前屏幕截图和用户任务。
请根据截图判断下一步应该执行什么操作，返回严格的 JSON 格式：

{
  "thought": "简短思考当前状态",
  "action": "click|type|hotkey|scroll|wait|done",
  "params": {
    // click: {"x": 整数, "y": 整数}
    // type: {"text": "要输入的文本"}
    // hotkey: {"keys": ["ctrl", "c"]}
    // scroll: {"x": 整数, "y": 整数, "clicks": 整数}
    // wait: {"seconds": 1}
    // done: {"result": "任务完成结果"}
  }
}

注意：
- 坐标是屏幕绝对像素坐标，必须根据截图估算
- 每次只返回一个 action
- 如果任务已完成，返回 action: "done"
- 操作后通常会伴随一次截图确认效果"""

    messages = [
        {"role": "system", "content": system_prompt},
    ]

    for h in history[-6:]:
        messages.append({
            "role": "user",
            "content": [
                {"type": "text", "text": h["step"]},
                {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{h['image']}"}},
            ]
        })
        messages.append({
            "role": "assistant",
            "content": json.dumps(h["action"], ensure_ascii=False)
        })

    messages.append({
        "role": "user",
        "content": [
            {"type": "text", "text": f"任务：{task}\n请根据当前截图返回下一步操作 JSON。"},
            {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{image_b64}"}},
        ]
    })

    return messages


def call_llm(api_key: str, messages: list, model: str = "kimi-k3") -> Optional[dict]:
    """调用 QuantumToken 多模态模型"""
    url = "https://trade.pianam.cn/api/v1/open/chat/completions"
    payload = {
        "model": model,
        "messages": messages,
        "request_id": f"agent-{int(time.time() * 1000)}",
        "temperature": 0.2,
    }
    headers = {
        "Content-Type": "application/json",
        "X-API-Key": api_key,
    }
    try:
        resp = requests.post(url, json=payload, headers=headers, timeout=60)
        resp.raise_for_status()
        data = resp.json()
        content = data["choices"][0]["message"]["content"]
        # 提取 JSON
        json_match = re.search(r"```json\s*(\{.*?\})\s*```", content, re.DOTALL)
        if json_match:
            content = json_match.group(1)
        else:
            json_match = re.search(r"(\{.*\})", content, re.DOTALL)
            if json_match:
                content = json_match.group(1)
        return json.loads(content)
    except Exception as e:
        print(f"调用 LLM 失败: {e}")
        traceback.print_exc()
        return None


def execute_action(action: dict) -> bool:
    """执行模型返回的动作"""
    action_type = action.get("action", "")
    params = action.get("params", {})

    try:
        if action_type == "click":
            x = int(params.get("x", 0))
            y = int(params.get("y", 0))
            pyautogui.click(x, y)
            print(f"  点击 ({x}, {y})")

        elif action_type == "type":
            text = params.get("text", "")
            pyautogui.typewrite(text, interval=0.01)
            print(f"  输入: {text}")

        elif action_type == "hotkey":
            keys = params.get("keys", [])
            pyautogui.hotkey(*keys)
            print(f"  快捷键: {'+'.join(keys)}")

        elif action_type == "scroll":
            x = int(params.get("x", 0))
            y = int(params.get("y", 0))
            clicks = int(params.get("clicks", -3))
            pyautogui.scroll(clicks, x, y)
            print(f"  滚动 ({x}, {y}) clicks={clicks}")

        elif action_type == "wait":
            seconds = float(params.get("seconds", 1))
            print(f"  等待 {seconds}s")
            time.sleep(seconds)

        elif action_type == "done":
            result = params.get("result", "任务完成")
            print(f"  ✅ {result}")
            return False

        else:
            print(f"  未知动作: {action_type}")

        return True
    except Exception as e:
        print(f"执行动作失败: {e}")
        traceback.print_exc()
        return True


def run_agent(api_key: str, task: str, max_steps: int = 20):
    """运行自动化任务"""
    print(f"🚀 开始任务: {task}")
    print(f"屏幕尺寸: {pyautogui.size()}")
    print(f"最多执行 {max_steps} 步，按 Ctrl+C 可中断\n")

    history = []
    for step in range(max_steps):
        print(f"--- 第 {step + 1} 步 ---")
        image_b64 = capture_screenshot()
        messages = build_messages(task, history, image_b64)
        decision = call_llm(api_key, messages)

        if decision is None:
            print("决策失败，重试...")
            time.sleep(2)
            continue

        print(f"思考: {decision.get('thought', '')}")
        print(f"动作: {json.dumps(decision.get('params', {}), ensure_ascii=False)}")

        history.append({
            "step": f"任务：{task} (第 {step + 1} 步)",
            "image": image_b64,
            "action": decision,
        })

        should_continue = execute_action(decision)
        if not should_continue:
            break

        time.sleep(1.5)
    else:
        print("\n⚠️ 达到最大步数限制，任务未完成")


def main():
    parser = argparse.ArgumentParser(description="QuantumToken 本地 AI 自动化 Agent")
    parser.add_argument("--api-key", required=True, help="QuantumToken API Key")
    parser.add_argument("--task", required=True, help="要执行的任务")
    parser.add_argument("--max-steps", type=int, default=20, help="最大执行步数")
    args = parser.parse_args()

    run_agent(args.api_key, args.task, args.max_steps)


if __name__ == "__main__":
    main()
