#!/usr/bin/env python3
"""陌生询盘四信号自查 —— 域名注册时间 · 官网是否空壳 · 发信 IP · 话术。

作者：朱强彬。只用 Python 标准库，Python 3.8+ 可跑，不用装任何东西。

用法：
  python3 inquiry_check.py --email buyer@example.com
  python3 inquiry_check.py --eml 询盘.eml            # 邮件另存为 .eml，四项一起查
  python3 inquiry_check.py --eml 询盘.eml --json     # 机器读的结果

结果是「信号」，不是结论：命中后先回信核实对方身份，不等于对方就是骗子。
"""
import argparse
import email
import ipaddress
import json
import re
import socket
import ssl
import sys
import urllib.error
import urllib.parse
import urllib.request
from datetime import datetime, timezone
from email import policy

UA = "Mozilla/5.0 (inquiry-check; +https://zhuqiangbin.com)"
TIMEOUT = 8
MAX_BYTES = 1_000_000

FREE_MAIL = {
    "gmail.com", "googlemail.com", "outlook.com", "hotmail.com", "live.com", "msn.com", "yahoo.com",
    "yahoo.co.uk", "ymail.com", "aol.com", "icloud.com", "me.com", "mail.com", "gmx.com", "gmx.de",
    "proton.me", "protonmail.com", "zoho.com", "yandex.com", "yandex.ru", "mail.ru", "qq.com",
    "163.com", "126.com", "sina.com", "sohu.com", "foxmail.com", "aliyun.com", "rediffmail.com",
}

PARKING = [
    "domain is for sale", "domain may be for sale", "buy this domain", "this domain is parked",
    "domain parking", "parked free", "parkingcrew", "sedoparking", "sedo.com", "dan.com", "afternic",
    "hugedomains", "bodis", "above.com", "domain for sale", "make an offer on this domain",
]
PLACEHOLDER = [
    "lorem ipsum", "coming soon", "under construction", "website is under construction",
    "welcome to nginx", "apache2 ubuntu default page", "it works!", "default web site page",
    "hello world!", "just another wordpress site", "site is not available",
]

MAIL_PROVIDERS = [
    "google", "microsoft", "outlook", "amazon", "zoho", "tencent", "alibaba", "aliyun", "netease",
    "263", "yahoo", "oath", "proofpoint", "mimecast", "mailgun", "sendgrid", "sparkpost", "mailchimp",
    "barracuda", "cisco", "fortinet", "sophos", "messagelabs", "symantec", "trend micro", "yandex",
    "mail.ru", "apple", "ovh mail",
]
HOSTING = [
    "hosting", "datacenter", "data center", "digitalocean", "linode", "akamai", "vultr", "hetzner",
    "ovh", "contabo", "choopa", "leaseweb", "m247", "colocrossing", "racknerd", "hostinger",
    "cloud", "server", "vps",
]
CONSUMER_ISP = [
    "broadband", "cable", "dsl", "fiber", "fibre", "mobile", "wireless", "cellular", "telecom",
    "telekom", "telecommunication", "comcast", "verizon", "at&t", "charter", "spectrum", "cox ",
    "vodafone", "orange", "jio", "airtel", "bsnl", "chinanet", "unicom", "china mobile", "turk telekom",
    "telefonica", "claro", "vivo", "tim ", "etisalat", "du ", "stc", "ptcl", "mtn", "globe", "pldt",
]

PITCH = [
    (r"\b(base|basic|existing|own)\s+design\b", "自带「基础设计」"),
    (r"\bwe\s+(already\s+)?have\s+(the\s+|a\s+|our\s+)?(design|drawings?)\b", "自称已有设计/图纸"),
    (r"\b(OEM|contract\s+manufactur\w*|manufacture\s+(it|them)\s+for\s+us)\b", "找工厂代工"),
    (r"\b(help|assist|support)\b[^.\n]{0,40}\b(CE|certificat\w*|approval|marking)\b", "请工厂帮办 CE / 认证"),
    (r"\b(CE)\s+(mark\w*|certificat\w*)\b", "提到 CE 认证"),
    (r"代工|贴牌", "找工厂代工"),
    (r"帮.{0,6}(办|申请|做).{0,6}(CE|认证|证书)", "请工厂帮办 CE / 认证"),
]

RED, YEL, OK, NA = "🔴", "🟡", "✅", "⚪"


# ── 工具函数 ──────────────────────────────────────────────────────────
class _SafeRedirect(urllib.request.HTTPRedirectHandler):
    """每一跳都重新核对目标是公网地址，防止被跳转到内网（SSRF）。"""
    max_redirections = 4

    def redirect_request(self, req, fp, code, msg, headers, newurl):
        host = urllib.parse.urlparse(newurl).hostname or ""
        if not is_public_host(host):
            raise urllib.error.URLError("redirect to non-public host blocked")
        return super().redirect_request(req, fp, code, msg, headers, newurl)


_OPENER = urllib.request.build_opener(
    _SafeRedirect, urllib.request.HTTPSHandler(context=ssl.create_default_context()))


def _get(url, max_bytes=MAX_BYTES):
    host = urllib.parse.urlparse(url).hostname or ""
    if not is_public_host(host):
        raise urllib.error.URLError("non-public host blocked")
    req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "*/*"})
    with _OPENER.open(req, timeout=TIMEOUT) as r:
        return r.geturl(), r.read(max_bytes).decode("utf-8", "replace")


def is_public_host(host):
    """防 SSRF：只允许解析到公网地址的主机。"""
    try:
        infos = socket.getaddrinfo(host, None)
    except OSError:
        return False
    for info in infos:
        ip = ipaddress.ip_address(info[4][0])
        if not ip.is_global:
            return False
    return True


def has_mx(domain):
    """域名有没有收信记录（不依赖第三方库：用系统 DNS 解析 MX 不方便，退而查 mail./www. 主机）。"""
    return any(is_public_host(h) for h in ("mail." + domain, "www." + domain))


def domain_of(addr):
    m = re.search(r"@([A-Za-z0-9.-]+\.[A-Za-z]{2,})", addr or "")
    return m.group(1).lower().strip(".") if m else None


def registrable(domain):
    """粗取主域名：mail.abc.co.uk → abc.co.uk；sales.abc.com → abc.com。"""
    parts = domain.split(".")
    if len(parts) >= 3 and len(parts[-1]) == 2 and parts[-2] in ("co", "com", "net", "org", "gov", "ac", "edu"):
        return ".".join(parts[-3:])
    return ".".join(parts[-2:])


# ── ① 域名注册时间 ────────────────────────────────────────────────────
def check_domain(addr):
    dom = domain_of(addr)
    if not dom:
        return {"id": 1, "name": "域名注册时间", "level": NA, "detail": "没有识别到发件邮箱"}
    if dom in FREE_MAIL:
        return {"id": 1, "name": "域名注册时间", "level": YEL,
                "detail": "对方用的是免费邮箱（%s），查不了公司域名。自称公司的，请对方改用公司邮箱确认" % dom}
    root = registrable(dom)
    try:
        _, body = _get("https://rdap.org/domain/" + root, 200_000)
        data = json.loads(body)
    except urllib.error.HTTPError as e:
        if e.code == 404 and not is_public_host(root) and not has_mx(root):
            return {"id": 1, "name": "域名注册时间", "level": RED, "detail": "%s 查不到注册记录，也解析不到，可能不存在或已过期" % root}
        if e.code == 404:
            return {"id": 1, "name": "域名注册时间", "level": NA, "detail": "%s 的后缀不支持在线查注册日期，可到 who.is 手查" % root}
        return {"id": 1, "name": "域名注册时间", "level": NA, "detail": "%s 注册信息暂时查不到（HTTP %s），可到 who.is 手查" % (root, e.code)}
    except (OSError, ValueError):
        return {"id": 1, "name": "域名注册时间", "level": NA, "detail": "%s 注册信息暂时查不到，可到 who.is 手查" % root}
    reg = next((ev.get("eventDate") for ev in data.get("events", []) if ev.get("eventAction") == "registration"), None)
    if not reg:
        return {"id": 1, "name": "域名注册时间", "level": NA, "detail": "%s 的注册日期未公开" % root}
    when = datetime.fromisoformat(reg.replace("Z", "+00:00"))
    days = (datetime.now(timezone.utc) - when).days
    if days < 183:
        lv, note = RED, "不到 6 个月"
    elif days < 365:
        lv, note = YEL, "不到 1 年"
    else:
        lv, note = OK, "约 %.1f 年" % (days / 365.25)
    return {"id": 1, "name": "域名注册时间", "level": lv,
            "detail": "%s 注册于 %s（%s）。新公司也会是新域名，要和其它几项一起看" % (root, when.date(), note),
            "domain": root, "age_days": days}


# ── ② 官网是否空壳 ────────────────────────────────────────────────────
def check_website(addr):
    dom = domain_of(addr)
    if not dom or dom in FREE_MAIL:
        return {"id": 2, "name": "官网", "level": NA, "detail": "没有公司域名，无法查官网"}
    root = registrable(dom)
    if not is_public_host(root) and not is_public_host("www." + root):
        return {"id": 2, "name": "官网", "level": RED, "detail": "%s 没有可访问的网站（域名解析不到）" % root}
    final, html = None, None
    for url in ("https://www." + root, "https://" + root, "http://www." + root, "http://" + root):
        host = re.sub(r"^https?://", "", url)
        if not is_public_host(host):
            continue
        try:
            final, html = _get(url)
            break
        except (OSError, ValueError):
            continue
    if html is None:
        return {"id": 2, "name": "官网", "level": NA,
                "detail": "%s 从服务器访问不到（部分网站会拦截机房访问），请用手机打开看看" % root}
    spa = len(re.findall(r"<script", html, re.I)) >= 3 and re.search(r'id=["\'](root|app|__next|__nuxt)["\']', html, re.I)
    low = html.lower()
    text = re.sub(r"<script.*?</script>|<style.*?</style>", " ", low, flags=re.S)
    text = re.sub(r"<[^>]+>", " ", text)
    text = re.sub(r"\s+", " ", text).strip()
    hits = []
    final_host = re.sub(r"^https?://([^/]+).*", r"\1", final or "").lower()
    if final_host and root not in final_host:
        hits.append("跳转到了别的网站（%s）" % final_host)
    if any(k in low for k in PARKING):
        hits.append("是域名停放 / 出售页")
    if any(k in low for k in PLACEHOLDER):
        hits.append("有占位或模板文字（coming soon / lorem ipsum 之类）")
    if spa and len(text) < 400:
        return {"id": 2, "name": "官网", "level": NA,
                "detail": "%s 是脚本渲染的网站，服务器读不到内容，请用浏览器打开看有没有地址电话" % root}
    if len(text) < 400:
        hits.append("正文很少（约 %d 字符）" % len(text))
    has_phone = bool(re.search(r"tel:|\+\d[\d\s().-]{7,}\d", low))
    has_addr = bool(re.search(r"address|street|road|avenue|building|floor|suite|p\.?o\.? box|地址|路\d|号楼", low))
    if not has_phone and not has_addr:
        hits.append("找不到电话和地址")
    if not hits:
        return {"id": 2, "name": "官网", "level": OK, "detail": "%s 有正常内容，并能找到联系方式" % root}
    lv = RED if any(("停放" in h or "跳转" in h) for h in hits) else YEL  # 🔴 只给硬证据
    return {"id": 2, "name": "官网", "level": lv, "detail": "%s：%s" % (root, "；".join(hits))}


# ── ③ 发信 IP ─────────────────────────────────────────────────────────
def first_hop_ip(msg):
    """从 Received 链里取最早一跳的公网 IP（最下面一条往上找）。"""
    cands = []
    for h in (msg.get_all("X-Originating-IP") or []):
        cands.append(h)
    received = msg.get_all("Received") or []
    for h in reversed(received):
        cands.append(h)
    for h in cands:
        for ip in re.findall(r"\[?(\d{1,3}(?:\.\d{1,3}){3})\]?", str(h)):
            try:
                if ipaddress.ip_address(ip).is_global:
                    return ip
            except ValueError:
                continue
    return None


def ip_lookup(ip):
    """IP 归属：ipinfo 为主，限流或失败换 ip-api.com。返回 {org, city, country} 或 None。"""
    try:
        _, body = _get("https://ipinfo.io/%s/json" % ip, 50_000)
        d = json.loads(body)
        if d.get("org"):
            return {"org": d.get("org"), "city": d.get("city"), "country": d.get("country")}
    except (OSError, ValueError):
        pass
    try:
        _, body = _get("http://ip-api.com/json/%s?fields=status,org,isp,as,city,country" % ip, 50_000)
        d = json.loads(body)
        if d.get("status") == "success":
            return {"org": " ".join(x for x in (d.get("as"), d.get("isp") or d.get("org")) if x),
                    "city": d.get("city"), "country": d.get("country")}
    except (OSError, ValueError):
        pass
    return None


def check_ip(msg):
    if msg is None:
        return {"id": 3, "name": "发信 IP", "level": NA, "detail": "没有提供邮件原文，跳过"}
    ip = first_hop_ip(msg)
    if not ip:
        return {"id": 3, "name": "发信 IP", "level": NA, "detail": "邮件原文里找不到发件人 IP（常见于网页版邮箱），此项跳过"}
    info = ip_lookup(ip)
    if not info:
        return {"id": 3, "name": "发信 IP", "level": NA, "detail": "%s 归属暂时查不到，可到 ipinfo.io 手查" % ip}
    org = (info.get("org") or "").lower()
    where = ", ".join(x for x in (info.get("city"), info.get("country")) if x)
    label = "%s（%s，%s）" % (ip, info.get("org") or "未知运营商", where or "未知地区")
    if any(k in org for k in MAIL_PROVIDERS):
        return {"id": 3, "name": "发信 IP", "level": OK, "detail": label + "：是邮件服务商的服务器，看不到发件人自己的网络"}
    if any(k in org for k in HOSTING):
        return {"id": 3, "name": "发信 IP", "level": YEL, "detail": label + "：机房 IP，VPN / 代理常见"}
    if any(k in org for k in CONSUMER_ISP):
        return {"id": 3, "name": "发信 IP", "level": YEL,
                "detail": label + "：电信运营商网络。公司域名邮箱从家庭或手机网络发出，要多留心（小公司用运营商宽带也正常）"}
    return {"id": 3, "name": "发信 IP", "level": OK, "detail": label}


# ── ④ 话术 ───────────────────────────────────────────────────────────
def check_pitch(body):
    if not body:
        return {"id": 4, "name": "话术", "level": NA, "detail": "没有邮件正文，跳过"}
    found = []
    for pat, label in PITCH:
        if re.search(pat, body, re.I) and label not in found:
            found.append(label)
    if not found:
        return {"id": 4, "name": "话术", "level": OK, "detail": "没发现「自带设计 + 找代工 + 帮办证」类说法"}
    lv = RED if len(found) >= 2 else YEL
    return {"id": 4, "name": "话术", "level": lv,
            "detail": "；".join(found) + "。真实项目买家一般会给项目名、最终用户、设计规范和交期"}


# ── 解析与汇总 ────────────────────────────────────────────────────────
def parse_raw(raw):
    """把粘贴的邮件原文 / .eml 解析成 (From 地址, Message, 正文)。"""
    if not raw or not raw.strip():
        return None, None, None
    msg = email.message_from_string(raw, policy=policy.default)
    if not msg.get("Received") and not msg.get("From"):
        return None, None, raw  # 只贴了正文
    frm = email.utils.parseaddr(str(msg.get("From") or ""))[1] or None
    body = ""
    try:
        part = msg.get_body(preferencelist=("plain", "html"))
        if part is not None:
            body = part.get_content()
            if part.get_content_type() == "text/html":
                body = re.sub(r"<[^>]+>", " ", body)
    except (KeyError, LookupError):
        body = ""
    return frm, msg, body


def run_checks(addr=None, raw=None):
    frm, msg, body = parse_raw(raw)
    addr = (addr or frm or "").strip()
    results = [check_domain(addr), check_website(addr), check_ip(msg), check_pitch(body)]
    reds = sum(r["level"] == RED for r in results)
    yels = sum(r["level"] == YEL for r in results)
    done = sum(r["level"] != NA for r in results)
    skipped = "；另有 %d 项没查到或没提供材料（⚪），要手查" % (4 - done) if done < 4 else ""
    if reds:
        verdict = "命中 %d 项 🔴：先回信核实身份（公司邮箱确认、项目与最终用户、规格书与交期），再决定要不要报价" % reds
    elif yels:
        verdict = "有 %d 项 🟡：可以正常往来，但报价前把这几项问清楚" % yels
    elif done:
        verdict = "已查的 %d 项没有发现明显信号" % done
    else:
        verdict = "四项都没能查成，请补充邮件原文或手查"
    verdict += skipped
    return {"email": addr, "results": results, "verdict": verdict,
            "note": "结果是信号，不是结论：命中不等于对方是骗子，没命中也不等于一定可靠。"}


def main():
    ap = argparse.ArgumentParser(description="陌生询盘四信号自查（作者：朱强彬）")
    ap.add_argument("--email", help="对方发件邮箱")
    ap.add_argument("--eml", help="邮件另存的 .eml 文件（含邮件头，可查四项）")
    ap.add_argument("--json", action="store_true", help="输出 JSON")
    a = ap.parse_args()
    raw = None
    if a.eml:
        with open(a.eml, "r", encoding="utf-8", errors="replace") as f:
            raw = f.read()
    if not a.email and not raw:
        ap.error("至少给 --email 或 --eml 其中一个")
    rep = run_checks(a.email, raw)
    if a.json:
        print(json.dumps(rep, ensure_ascii=False, indent=2))
        return
    print("\n陌生询盘四信号自查：%s\n" % (rep["email"] or "（未识别邮箱）"))
    for r in rep["results"]:
        print("%s  %s：%s" % (r["level"], r["name"], r["detail"]))
    print("\n→ " + rep["verdict"])
    print("  " + rep["note"] + "\n")


if __name__ == "__main__":
    main()
