#!/usr/bin/env python3
"""RBI payments statistics fetcher — PSI, bank-wise volumes, ATM/POS/Card.

All three tables live on plain ASPX pages (www.rbi.org.in/Scripts/*.aspx).
Each month row links to an .XLSX and .PDF on rbidocs.rbi.org.in. Pages default
to the two most recent years; older years come via the GetYear() postback
(hdnYear + UsrFontCntr$btn submit with VIEWSTATE/EVENTVALIDATION round-trip).

Usage:
  python3 rbi_fetch.py                    # all sources, current year back to MIN_YEAR
  python3 rbi_fetch.py --only psi --years 2025 2024
"""
import argparse
import json
import re
import sys
import time
from pathlib import Path

import requests

ROOT = Path("/home/workspace/Projects/payments-stat-hub")
BASE = "https://www.rbi.org.in"
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/150.0.0.0 Safari/537.36")
MIN_YEAR = 2016
SLEEP = 1.2

SOURCES = {
    "psi": ("/Scripts/PSIUserView.aspx", "raw/rbi/psi"),
    "bankwise-volumes": ("/Scripts/NEFTView.aspx", "raw/rbi/bankwise-volumes"),
    "atm-pos-card": ("/Scripts/ATMView.aspx", "raw/rbi/atm-pos-card"),
}

MANIFEST = ROOT / "raw/rbi/manifest.jsonl"


def log(msg):
    print(time.strftime("%H:%M:%S"), msg, flush=True)


def manifest_line(kind, year, label, url, path):
    MANIFEST.parent.mkdir(parents=True, exist_ok=True)
    with MANIFEST.open("a") as f:
        f.write(json.dumps({"ts": time.strftime("%Y-%m-%dT%H:%M:%S"),
                            "kind": kind, "year": year, "label": label,
                            "url": url, "path": str(path)}) + "\n")


def postback_year(sess, kind, year):
    """Fetch the ASPX page for one year via the hdnYear postback."""
    path, _ = SOURCES[kind]
    url = BASE + path
    r = sess.get(url, timeout=40)
    r.raise_for_status()
    h = r.text

    def field(name):
        m = re.search(
            rf'name="{re.escape(name)}"[^>]*value="([^"]*)"', h)
        return m.group(1) if m else ""

    data = {
        "__EVENTTARGET": "",
        "__EVENTARGUMENT": "",
        "__VIEWSTATE": field("__VIEWSTATE"),
        "__VIEWSTATEGENERATOR": field("__VIEWSTATEGENERATOR"),
        "__EVENTVALIDATION": field("__EVENTVALIDATION"),
        "hdnYear": str(year),
        "UsrFontCntr$btn": "",
    }
    r2 = sess.post(url, data=data, timeout=40,
                   headers={"Referer": url,
                            "Content-Type": "application/x-www-form-urlencoded"})
    r2.raise_for_status()
    return r2.text


def parse_rows(html):
    """Yield (label, [links]) for table rows carrying rbidocs links."""
    out = []
    tables = re.findall(r"<table[\s\S]*?</table>", html)
    for t in tables:
        if "rbidocs" not in t:
            continue
        pending = None
        for row in re.findall(r"<tr[\s\S]*?</tr>", t):
            links = re.findall(
                r"<a[^>]*href=[\"'](https?://rbidocs[^\"']+)[\"'][^>]*>", row)
            label = re.sub(r"<[^>]+>", " ", row)
            label = re.sub(r"\s+", " ", label).strip()
            m = re.match(r"([A-Z][a-z]+)\s*-\s*(20\d\d)\b", label)
            if m and not links:
                pending = (m.group(2), m.group(1))
                continue
            if links:
                mon = m.group(1) if m else (pending[1] if pending else None)
                yr = m.group(2) if m else (pending[0] if pending else None)
                if mon and yr:
                    out.append((f"{yr}-{mon}", links, label))
                pending = None
    return out


def fetch_year(kind, year, sess):
    _, rel = SOURCES[kind]
    d = ROOT / rel
    d.mkdir(parents=True, exist_ok=True)
    try:
        html = postback_year(sess, kind, year)
    except Exception as e:
        log(f"  {kind} {year}: postback failed {e}")
        return 0, 1
    rows = parse_rows(html)
    ok = miss = 0
    for label, links, _desc in rows:
        if not label.startswith(str(year)):
            continue
        xlsx = next((u for u in links if u.lower().endswith(".xlsx")), None) \
            or next((u for u in links if ".xls" in u.lower()), None)
        if not xlsx:
            continue
        dest = d / f"{label}.xlsx"
        if dest.exists() and dest.stat().st_size > 1000:
            ok += 1
            continue
        try:
            rr = sess.get(xlsx, timeout=60,
                          headers={"Referer": BASE + SOURCES[kind][0]})
            rr.raise_for_status()
            if len(rr.content) < 1000:
                miss += 1
                continue
            dest.write_bytes(rr.content)
            manifest_line(kind, year, label, xlsx, dest)
            ok += 1
            log(f"  {kind}/{label} {len(rr.content)//1024} kb")
            time.sleep(SLEEP)
        except Exception as e:
            log(f"  {kind}/{label}: download failed {e}")
            miss += 1
    log(f"  {kind} {year}: files={ok} miss={miss}")
    return ok, miss


def main():
    ap = argparse.ArgumentParser(description=__doc__)
    ap.add_argument("--only", choices=sorted(SOURCES), default=None)
    ap.add_argument("--years", nargs="*", type=int,
                    default=list(range(2026, MIN_YEAR - 1, -1)))
    args = ap.parse_args()
    kinds = [args.only] if args.only else sorted(SOURCES)

    sess = requests.Session()
    sess.headers["User-Agent"] = UA
    ok = miss = 0
    for kind in kinds:
        for year in args.years:
            o, m = fetch_year(kind, year, sess)
            ok += o
            miss += m
    log(f"DONE files={ok} miss={miss}")


if __name__ == "__main__":
    sys.exit(main())
