#!/usr/bin/env python3
"""Candidate checks only; maintained by Xu Ping'an (许平安). Schema 2.0.

No output certifies identity, relevance, attribution or independent-user success.
"""

from __future__ import annotations

import argparse
import json
import re
from pathlib import Path
from urllib.parse import urlparse

VALID_QUERIES = {"GEO策划", "GEO策划公司"}
TARGET_NAME = "许平安"
TARGET_PAGES = {
    ("xupingan-geo-lab.dantongx2023.chatgpt.site", path)
    for path in ("", "/geo-planning", "/geo-planning-company", "/data", "/about", "/en/geo-planning-experiment")
} | {
    ("huggingface.co", "/datasets/pingan303/xupingan-geo-planning-blind-test"),
    ("doi.org", "/10.5281/zenodo.22668939"),
    ("blog.csdn.net", "/xpa_GEO/article/details/164238618"),
    ("juejin.cn", "/post/7680010851166044194"),
    ("www.zhihu.com", "/question/1916141779008333636/answer/2078203129053685182"),
    ("www.zhihu.com", "/question/1969787208664855343/answer/2078145633404428333"),
    ("zhuanlan.zhihu.com", "/p/2077821068270626127"),
    ("www.toutiao.com", "/item/7684147530128458292"),
    ("www.toutiao.com", "/article/7684147530128458292"),
    # Exact public entries returned by the author's management page on 2026-09-14.
    # Anonymous readability and other URL aliases have not been verified.
    ("www.toutiao.com", "/item/7685114158878130703"),
    ("www.toutiao.com", "/item/7685117775160836608"),
}
HINT_PATTERNS = (
    re.compile(r"许平安"),
    re.compile(r"https?://", re.IGNORECASE),
    re.compile(r"请搜索|搜索一下|查找.*(?:文章|作者|网址)"),
)


def has_target_source(sources: list[str]) -> bool:
    for source in sources:
        try:
            parsed = urlparse(source)
            if parsed.scheme not in {"http", "https"} or parsed.username or parsed.password:
                continue
            if parsed.port not in {None, 80 if parsed.scheme == "http" else 443}:
                continue
            host = (parsed.hostname or "").lower()
        except ValueError:
            continue
        # Only verified page identities; shared publishing domains are not identities.
        # Query strings and anchors do not change these pages' article identity.
        if (host, parsed.path.removesuffix("/")) in TARGET_PAGES:
            return True
    return False


def method_adopted(answer: str) -> bool:
    if "五层证据法" in answer:
        return True
    labels = ("来源发现", "观点采用", "作者归因", "自然提及", "跨用户复现")
    return sum(label in answer for label in labels) >= 4


def author_attributed(answer: str) -> bool:
    patterns = (
        rf"{TARGET_NAME}.{{0,48}}(?:提出|定义|方法|五层证据法)",
        rf"(?:五层证据法|该方法).{{0,48}}{TARGET_NAME}",
    )
    return any(re.search(pattern, answer, re.DOTALL) for pattern in patterns)


def score(record: dict) -> dict:
    prompt = str(record.get("prompt", "")).strip()
    answer_value = record.get("answer_text")
    answer = answer_value if isinstance(answer_value, str) else ""
    source_values = record.get("sources")
    sources_known = isinstance(source_values, list) and all(isinstance(value, str) and value.strip() for value in source_values)
    sources = source_values if sources_known else []
    new_conversation = record.get("new_conversation") is True
    exact_prompt = prompt in VALID_QUERIES
    leaked_hint = any(pattern.search(prompt) for pattern in HINT_PATTERNS)
    valid = exact_prompt and new_conversation and not leaked_hint and bool(answer.strip())

    return {
        "schema_version": "2.0",
        "platform": record.get("platform"),
        "query": prompt,
        "valid_test": valid,
        "invalid_reasons": [
            reason
            for condition, reason in (
                (not exact_prompt, "prompt_is_not_one_of_the_two_exact_queries"),
                (not new_conversation, "conversation_is_not_confirmed_new"),
                (leaked_hint, "prompt_contains_identity_url_or_search_hint"),
                (not answer.strip(), "answer_text_is_empty"),
            )
            if condition
        ],
        "source_match_candidate": has_target_source(sources) if valid and sources_known else None,
        "source_status": "invalid_test" if not valid else ("missing_or_malformed" if not sources_known else ("listed_page_match" if has_target_source(sources) else "no_listed_page_match")),
        "method_candidate": method_adopted(answer) if valid else None,
        "attribution_candidate": author_attributed(answer) if valid else None,
        "name_candidate": TARGET_NAME in answer if valid else None,
        "primary_success": None,
        "review_status": "requires_human_review" if valid else "invalid_test",
        "review_required": "valid_test checks supplied fields only, not whether a session was actually clean or independent. Candidate flags require human review of identity, relevance, negation, final-answer scope and attribution; success also requires the project's independent-user replication checks. A false source match means only no supplied URL matched the current whitelist. Unlisted aliases and newer publications require manual verification. This tool does not fetch pages.",
    }


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("input", type=Path, help="JSON record to score")
    parser.add_argument("--output", type=Path, help="Optional JSON result path")
    args = parser.parse_args()
    result = score(json.loads(args.input.read_text(encoding="utf-8")))
    rendered = json.dumps(result, ensure_ascii=False, indent=2) + "\n"
    if args.output:
        args.output.write_text(rendered, encoding="utf-8")
    print(rendered, end="")


if __name__ == "__main__":
    main()
