#!/usr/bin/env python3
"""
Test different thinking modes on GLM-4.7-flash.
Note: Flash models may not support thinking - this tests what's available.
"""

import os
import sys
import time

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))

from core.zai_client import ZAiClient


def test_thinking_modes(api_key: str):
    """Test all thinking mode combinations."""

    print("=" * 60)
    print("Testing Thinking Modes on GLM-4.7-flash")
    print("=" * 60 + "\n")

    test_prompt = "What is 15 * 23? Show your work briefly."

    configs = [
        {"name": "Thinking OFF", "thinking": {"type": "disabled"}},
        {"name": "Thinking ON", "thinking": {"type": "enabled"}},
        {
            "name": "Thinking + Preserved",
            "thinking": {"type": "enabled", "clear_thinking": False},
        },
    ]

    with ZAiClient(api_key, timeout=30.0) as client:
        for config in configs:
            print(f"\n--- {config['name']} ---")

            try:
                start = time.time()

                response = client.chat_completion(
                    model="glm-4.7-flash",
                    messages=[{"role": "user", "content": test_prompt}],
                    thinking=config["thinking"],
                    max_tokens=500,
                    temperature=0.7,
                )

                latency = (time.time() - start) * 1000

                print(f"Latency: {latency:.0f}ms")
                print(f"Tokens: {response.usage}")

                if response.reasoning_content:
                    print(f"\n[Reasoning ({len(response.reasoning_content)} chars)]:")
                    print(
                        response.reasoning_content[:300]
                        + ("..." if len(response.reasoning_content) > 300 else "")
                    )

                print(f"\n[Response]:")
                print(response.content)

            except Exception as e:
                print(f"[ERROR] {e}")

            time.sleep(2)

    print("\n" + "=" * 60)
    print("Test Complete")
    print("=" * 60)


def test_streaming(api_key: str):
    """Test streaming with reasoning capture."""

    print("\n" + "=" * 60)
    print("Testing Streaming Mode")
    print("=" * 60 + "\n")

    with ZAiClient(api_key, timeout=30.0) as client:
        print("Prompt: Count from 1 to 5 slowly.\n")

        reasoning_total = ""
        content_total = ""

        try:
            for reasoning, content in client.chat_stream(
                model="glm-4.7-flash",
                messages=[{"role": "user", "content": "Count from 1 to 5 slowly."}],
                thinking={"type": "enabled"},
                max_tokens=100,
            ):
                if reasoning:
                    reasoning_total += reasoning
                    print(f"\033[90m[{reasoning}]\033[0m", end="", flush=True)
                if content:
                    content_total += content
                    print(content, end="", flush=True)

            print("\n")
            print(f"Reasoning captured: {len(reasoning_total)} chars")
            print(f"Content: {len(content_total)} chars")

        except Exception as e:
            print(f"[ERROR] {e}")


def interactive_session(api_key: str):
    """Interactive multi-turn session with preserved thinking."""

    print("\n" + "=" * 60)
    print("Interactive Session (type 'quit' to exit)")
    print("=" * 60 + "\n")

    history = []
    thinking_enabled = True
    preserved = True

    with ZAiClient(api_key, timeout=60.0) as client:
        while True:
            try:
                user_input = input("You> ").strip()
                if not user_input:
                    continue
                if user_input.lower() in ("quit", "exit"):
                    break

                thinking_config = None
                if thinking_enabled:
                    thinking_config = {
                        "type": "enabled",
                        "clear_thinking": not preserved,
                    }

                messages = history.copy()
                messages.append({"role": "user", "content": user_input})

                print("\n[GLM] ", end="", flush=True)

                reasoning_parts = []
                content_parts = []

                for reasoning, content in client.chat_stream(
                    model="glm-4.7-flash",
                    messages=messages,
                    thinking=thinking_config,
                    max_tokens=1000,
                ):
                    if reasoning:
                        reasoning_parts.append(reasoning)
                    if content:
                        content_parts.append(content)
                        print(content, end="", flush=True)

                print()

                full_reasoning = "".join(reasoning_parts)
                full_content = "".join(content_parts)

                if full_reasoning:
                    print(f"\n\033[90m[Thinking: {len(full_reasoning)} chars]\033[0m")

                if preserved and full_reasoning:
                    history.append({"role": "user", "content": user_input})
                    history.append(
                        {
                            "role": "assistant",
                            "content": full_content,
                            "reasoning_content": full_reasoning,
                        }
                    )
                else:
                    history.append({"role": "user", "content": user_input})
                    history.append({"role": "assistant", "content": full_content})

            except KeyboardInterrupt:
                print("\n[Interrupted]")
            except Exception as e:
                print(f"[ERROR] {e}")


if __name__ == "__main__":
    api_key = os.environ.get("ZAI_API_KEY")

    if not api_key:
        print("ERROR: ZAI_API_KEY not set")
        sys.exit(1)

    if len(sys.argv) > 1:
        mode = sys.argv[1]
        if mode == "modes":
            test_thinking_modes(api_key)
        elif mode == "stream":
            test_streaming(api_key)
        elif mode == "chat":
            interactive_session(api_key)
        else:
            print(f"Unknown mode: {mode}")
            print("Usage: python test_thinking.py [modes|stream|chat]")
    else:
        print("SAL-v4 Thinking Mode Tests")
        print("=" * 40)
        print("  modes  - Test different thinking configs")
        print("  stream - Test streaming with reasoning")
        print("  chat   - Interactive multi-turn session")
        print()
        print("Usage: ZAI_API_KEY=key python test_thinking.py <mode>")
