#!/usr/bin/env python3
"""glance: standalone image description, Q&A, and OCR CLI."""

import argparse
from pathlib import Path
import sys

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from vision_client import (  # noqa: E402
    VisionError,
    describe_image,
    image_path_to_data_url,
    load_default_env,
)


def region_data_url(path, region):
    try:
        from PIL import Image
    except ImportError:
        raise VisionError("--region requires Pillow; install the optional dependency pillow first")
    try:
        x1, y1, x2, y2 = (int(v) for v in region.split(","))
    except ValueError:
        raise VisionError("--region expects four integers: X1,Y1,X2,Y2 (pixels)")
    import base64
    import io
    try:
        with Image.open(Path(path).expanduser()) as image:
            width, height = image.size
            box = (max(0, min(x1, x2)), max(0, min(y1, y2)),
                   min(width, max(x1, x2)), min(height, max(y1, y2)))
            if box[2] <= box[0] or box[3] <= box[1]:
                raise VisionError(f"--region {region} is empty after clamping to {width}x{height}")
            buffer = io.BytesIO()
            image.crop(box).save(buffer, format="PNG")
    except (OSError, ValueError) as exc:
        raise VisionError(f"Cannot read image: {path}") from exc
    return "data:image/png;base64," + base64.b64encode(buffer.getvalue()).decode()


def build_prompt(args, count):
    if args.ocr is not None:
        extra = f" Additional requirements: {args.ocr}" if args.ocr else ""
        scope = "these images" if count > 1 else "this image"
        return (
            f"Transcribe every piece of visible text in {scope} verbatim (titles, body text, labels, watermarks, etc.), "
            "line by line, without omitting any characters. Do not rewrite, summarize, or translate the text, "
            "and do not add any preamble, explanation, or extra content."
            + (" Label each image's text with its ordinal (Image 1, Image 2, ...)." if count > 1 else "")
            + extra
        )
    if args.query:
        return args.query
    if count > 1:
        return ("Describe each image in detail (label them Image 1, Image 2, ...), "
                "then point out the notable differences between them.")
    return None


def main():
    parser = argparse.ArgumentParser(
        prog="glance",
        description="Describe, answer questions about, or OCR an image with the configured vision model",
    )
    parser.add_argument("images", nargs="+", metavar="image",
                        help="path(s) to the image(s); pass several to compare them in one call")
    group = parser.add_mutually_exclusive_group()
    group.add_argument("-q", "--query", help="ask a question about the image(s)")
    group.add_argument("--ocr", nargs="?", const="", metavar="EXTRA", help="transcribe all visible text verbatim")
    parser.add_argument("--region", metavar="X1,Y1,X2,Y2",
                        help="crop to this pixel box (e.g. from ground) and send only the crop")
    args = parser.parse_args()
    load_default_env()
    try:
        if args.region and len(args.images) > 1:
            raise VisionError("--region works with exactly one image")
        urls = ([region_data_url(args.images[0], args.region)] if args.region
                else [image_path_to_data_url(path) for path in args.images])
        answer = describe_image(
            urls,
            build_prompt(args, len(urls)),
            max_tokens=None,  # no output cap; the vision model decides
            apply_lang=args.ocr is None,
        )
    except VisionError as exc:
        parser.exit(1, f"glance: {exc}\n")
    print(answer)


if __name__ == "__main__":
    main()
