import asyncio
import json

import pytest

from proxyscraper import sources as srcs

DAY = srcs.DAY


def test_normalize_github_urls():
    assert (
        srcs.normalize_url("https://github.com/a/b/raw/refs/heads/main/http.txt,,ColonURL")
        == "https://raw.githubusercontent.com/a/b/main/http.txt"
    )
    assert (
        srcs.normalize_url("https://raw.githubusercontent.com/a/b/refs/heads/main/x.txt")
        == "https://raw.githubusercontent.com/a/b/main/x.txt"
    )
    assert srcs.normalize_url("https://x.org/{YYYY}/{MM}.txt") is None
    assert srcs.normalize_url("not a url") is None


def test_parse_url_list_skips_comments_and_dedupes():
    data = b"# comment\nhttps://x.org/a.txt\nhttps://x.org/a.txt\n\nhttps://x.org/b.txt,,SpaceURL\n"
    assert srcs.parse_url_list(data, "https") == {"https://x.org/a.txt": "http", "https://x.org/b.txt": "http"}


def test_parse_monosans_toml_sections():
    toml = b"""
[scraping]
enabled = true
[scraping.http]
urls = [
  "https://h.org/http.txt",
  # "https://h.org/disabled.txt",
]
[scraping.socks5]
urls = ["https://h.org/s5.txt"]
[output]
path = "https://not-a-source.org/x"
"""
    assert srcs.parse_monosans_toml(toml) == {"https://h.org/http.txt": "http", "https://h.org/s5.txt": "socks5"}


def test_classify_path():
    assert srcs.classify_path("proxies/protocols/socks5/data.txt") == "socks5"
    assert srcs.classify_path("https.txt") == "http"
    assert srcs.classify_path("all_proxies.txt") == "auto"
    assert srcs.classify_path("countries/DE/http.txt") is None
    assert srcs.classify_path("v2ray/vmess.txt") is None
    assert srcs.classify_path("us.txt") is None
    assert srcs.classify_path("http.json") is None
    assert srcs.classify_path("a/b/c/d/http.txt") is None


def test_resolve_meta_tolerates_failures():
    async def get(url, timeout=0, headers=None):
        if "bad" in url:
            raise ConnectionError
        return b"https://x.org/one.txt\n"

    meta = [{"url": "https://ok", "type": "socks4"}, {"url": "https://bad", "type": "http"}]
    found, ok = asyncio.run(srcs.resolve_meta(meta, get))
    assert (found, ok) == ({"https://x.org/one.txt": "socks4"}, 1)


def test_discover_github_filters_spam_owners_and_files():
    repos = [{"full_name": f"spam/r{i}", "default_branch": "main", "stargazers_count": 100 - i} for i in range(5)]
    repos.append({"full_name": "good/list", "default_branch": "main", "stargazers_count": 1})
    tree = {"tree": [
        {"type": "blob", "path": "http.txt", "size": 5000},
        {"type": "blob", "path": "README.md", "size": 5000},
        {"type": "blob", "path": "socks5.txt", "size": 10},  # too small
    ]}

    async def get(url, timeout=0, headers=None):
        if "search/repositories" in url:
            return json.dumps({"items": repos}).encode()
        return json.dumps(tree).encode()

    found = asyncio.run(srcs.discover_github(get, token=None, max_repos=50))
    owners = {u.split("/")[3] for u in found}
    assert owners == {"spam", "good"}
    assert sum(1 for u in found if "/spam/" in u) == srcs.MAX_REPOS_PER_OWNER
    assert all(u.endswith("/http.txt") and t == "http" for u, t in found.items())


def test_stats_skip_rules(tmp_path):
    st = srcs.SourceStats(tmp_path / "s.json")
    now = 1_000_000.0

    # unreachable after 3 failures, allowed again after the pause
    for _ in range(3):
        st.record_fetch("u", None, 0, now=now)
    assert st.skip_reason("u", now=now + 1) == "unreachable"
    assert st.skip_reason("u", now=now + srcs.UNREACHABLE_PAUSE + 1) is None

    # outdated: content unchanged for more than a week
    st.record_fetch("s", b"same", 10, now=now)
    st.record_fetch("s", b"same", 10, now=now + 8 * DAY)
    assert st.skip_reason("s", now=now + 8 * DAY) == "outdated"
    st.record_fetch("s", b"new", 10, now=now + 9 * DAY)
    assert st.skip_reason("s", now=now + 9 * DAY) is None

    # dead: two runs, enough checks, no hit
    st.record_checks({"d": (400, 0)})
    assert st.skip_reason("d") is None  # one run isn't enough
    st.record_checks({"d": (400, 0)})
    assert st.skip_reason("d") == "dead"


def test_stats_score_and_roundtrip(tmp_path):
    path = tmp_path / "s.json"
    st = srcs.SourceStats(path)
    st.record_checks({"good": (100, 40), "bad": (100, 1)})
    assert st.score("good") > st.score("bad") > 0
    assert 0 < st.score("unknown") < 0.05
    st.save()
    again = srcs.SourceStats(path)
    assert again.score("good") == st.score("good")


def test_stats_corrupt_file_starts_fresh(tmp_path):
    path = tmp_path / "s.json"
    path.write_text("{kaputt")
    with pytest.warns(UserWarning, match="unreadable"):
        assert srcs.SourceStats(path).records == {}


def test_curated_sources_file_is_valid():
    sources, meta = srcs.load_source_file()
    assert len(sources) > 100
    assert set(sources.values()) <= set(srcs.SOURCE_TYPES)
    assert meta and all("url" in m for m in meta)


def test_stats_with_fields_from_another_version_are_still_read(tmp_path):
    path = tmp_path / "s.json"
    path.write_text(json.dumps({"u": {"first_seen": 1.0, "count": 5, "some_future_field": "x"}}))
    st = srcs.SourceStats(path)
    assert st.get("u").count == 5


def test_broken_entries_in_the_stats_file_dont_crash(tmp_path):
    path = tmp_path / "s.json"
    path.write_text(json.dumps({"u": None, "v": [1], "w": {"count": 3}}))
    st = srcs.SourceStats(path)
    assert st.get("w").count == 3 and "u" not in st.records


@pytest.mark.parametrize("mirror", [
    "https://cdn.jsdelivr.net/gh/proxifly/free-proxy-list@main/proxies/all/data.txt",
    "https://fastly.jsdelivr.net/gh/proxifly/free-proxy-list@main/proxies/all/data.txt",
    "https://raw.githack.com/proxifly/free-proxy-list/main/proxies/all/data.txt",
    "https://rawcdn.githack.com/proxifly/free-proxy-list/main/proxies/all/data.txt",
    "https://github.com/proxifly/free-proxy-list/blob/main/proxies/all/data.txt",
])
def test_mirrors_of_a_github_file_are_the_same_source(mirror):
    assert srcs.normalize_url(mirror) == \
        "https://raw.githubusercontent.com/proxifly/free-proxy-list/main/proxies/all/data.txt"


def test_jsdelivr_without_a_branch_is_left_alone():
    # @latest-style or no ref: which branch it is isn't clear, so it stays as it is
    url = "https://cdn.jsdelivr.net/gh/someone/list/proxies.txt"
    assert srcs.normalize_url(url) == url


@pytest.mark.parametrize("url", [
    "https://cdn.jsdelivr.net/gh/someone/list@latest/proxies.txt",
    "https://cdn.jsdelivr.net/gh/someone/list@1/proxies.txt",
    "https://cdn.jsdelivr.net/gh/someone/list@2.3/proxies.txt",
    "https://cdn.jsdelivr.net/gh/someone/list@v1.2.0/proxies.txt",
])
def test_jsdelivr_versions_are_not_github_branches(url):
    # jsDelivr resolves these itself – on raw.githubusercontent.com they'd be a 404
    assert srcs.normalize_url(url) == url
