#!/usr/bin/env python3
"""Backfill the complete per-stake HEX event history from Dwellir archives.

PulseChain's HEX history = the shared Ethereum era (HEX deploy -> block
17,232,999, served ONLY by the Ethereum host — the PLS archive holds zero
pre-fork logs) + the PulseChain era (17,233,000 -> tip, PLS host).  The fork
boundary was pinned by block-hash bisection 2026-08-30.

One eth_getLogs per range fetches all four stake events together (topics[0]
as an OR-list).  Ranges self-shrink: Dwellir's over-limit errors name the
exact retry range, which we follow; anything else halves.  Output is decoded
JSONL per (era, event) under data/hex-stakes/, with a checkpoint file per era
so a killed run resumes where it stopped.  Raw values stay exact strings
(wei/hearts) — the unrounded-data law applies to archives too.
"""
import json, os, re, sys, time, urllib.request

ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
OUT = os.path.join(ROOT, 'data', 'hex-stakes')
os.makedirs(OUT, exist_ok=True)

# the paid door: PLS_RPC_PAID in the robot (the same secret the census uses),
# the key file on the Mac — 2026-09-09, when the map went onto a daily robot
raw = (os.environ.get('PLS_RPC_PAID') or '').strip() or open(os.path.expanduser('~/.dwellir-pls.url')).read().strip()
KEY = re.match(r'https://[^/]+/(.+)', raw).group(1)
PLS_URL = raw
ETH_URL = f'https://api-ethereum-mainnet.n.dwellir.com/{KEY}'

HEX = '0x2b591e99afE9f32eAA6214f7B7629768c40Eeb39'
FORK_FIRST_PLS = 17_233_000          # first PulseChain-only block
ETH_SCAN_FROM = 9_000_000            # HEX deployed ~9,041,184; sparse lead-in is cheap

TOPICS = {
    '0x14872dc760f33532684e68e1b6d5fd3f71ba7b07dee76bdb2b084f28b74233ef': 'stakeStart',
    '0x72d9c5a7ab13846e08d9c838f9e866a1bb4a66a2fd3ba3c9e7da3cf9e394dfd7': 'stakeEnd',
    '0xd824970a2cf19cc2b630c87ce5b00f67301cac3ac60513d027c7a39129f93b46': 'stakeGoodAccounting',
    # ShareRateChange(uint256,uint40) — hash proven 2026-08-30 from a live receipt
    # matched against the subgraph.  (0xb8d6eb54… , the first guess, is
    # DailyDataUpdate — same 2-topic/1-word shape, and the discovery window
    # happened to contain only that one.  The comparison's 0/100 SRC match
    # caught it.)
    '0x9861fa0ed101659f7a59b4583fcc798dfa4f3b419bea371c8ee2ad0ffe13a31e': 'shareRateChange',
}

# ---- the HSI manager's own events (2026-09-02, BossA: "add the hsi data set") ----
# Hedron's HEX Stake Instance Manager, the registry of every HSI; the same
# address on both chains.  Five events, each with the block timestamp as its
# one data word (verified in the deployed source, knowledge-base/sources/
# HEXStakeInstanceManager.sol).  HSITransfer is the liquidation-auction
# hand-over — the only way a plain (untokenized) instance changes hands;
# see knowledge-base/hsi-transfer-liquidation-handover.md.
HSIM = '0x8bd3d1472a656e312e94fb1bbdd599b8c51d18e3'
HSIM_SCAN_FROM = 14_278_000          # first HSIStart at 14,278,406 (Ethereum, Feb 2022)
TOPICS_HSIM = {
    '0xd680a9b62662668ffed760ca1d0741736980d08c278efca9e0c6dcc1a4c166ca': 'hsiStart',       # HSIStart(uint256,address,address)
    '0x14d0fe09f225917f351bd3b122714cfcc1c45015a67232167d4b561e186b26de': 'hsiEnd',         # HSIEnd(uint256,address,address)
    '0xc24b27b33d05d2d17b1cf97ccbe0c85b21236ad0f06ba8359bf46f8e3b2749b6': 'hsiTransfer',    # HSITransfer(uint256,address,address,address)
    '0xed10b8f4c54a638850d395c632b529baa72a9c68ee7ed868f15a0468405d5147': 'hsiTokenize',    # HSITokenize(uint256,uint256,address,address)
    '0x6bce622a5976965d5b72e030c8cab9696faae9e320a35bfd263b1681ae7f2490': 'hsiDetokenize',  # HSIDetokenize(uint256,uint256,address,address)
}

def decode_hsim(kind, log):
    t = log['topics']
    row = {'b': int(log['blockNumber'], 16), 'li': int(log['logIndex'], 16), 'ts': int(log['data'][2:66], 16)}
    if kind in ('hsiStart', 'hsiEnd'):
        row.update(inst='0x' + t[1][-40:], staker='0x' + t[2][-40:])
    elif kind == 'hsiTransfer':
        row.update(inst='0x' + t[1][-40:], **{'from': '0x' + t[2][-40:]}, to='0x' + t[3][-40:])
    else:  # hsiTokenize / hsiDetokenize
        row.update(tid=int(t[1], 16), inst='0x' + t[2][-40:], staker='0x' + t[3][-40:])
    return row

def rpc(url, method, params, timeout=75):
    body = json.dumps({'jsonrpc': '2.0', 'id': 1, 'method': method, 'params': params}).encode()
    req = urllib.request.Request(url, body, {'Content-Type': 'application/json'})
    return json.load(urllib.request.urlopen(req, timeout=timeout))

def bits(x, shift, width):
    return (x >> shift) & ((1 << width) - 1)

def decode(kind, log):
    t = log['topics']
    d0 = int(log['data'][2:66], 16)
    row = {'b': int(log['blockNumber'], 16), 'li': int(log['logIndex'], 16)}
    if kind in ('stakeStart', 'stakeEnd', 'stakeGoodAccounting'):
        row['addr'] = '0x' + t[1][-40:]
        row['id'] = int(t[2], 16)
    if kind == 'stakeStart':
        row.update(ts=bits(d0, 0, 40), hearts=str(bits(d0, 40, 72)),
                   shares=str(bits(d0, 112, 72)), days=bits(d0, 184, 16),
                   auto=bits(d0, 200, 1))
    elif kind in ('stakeEnd', 'stakeGoodAccounting'):
        d1 = int(log['data'][66:130], 16)
        row.update(ts=bits(d0, 0, 40), hearts=str(bits(d0, 40, 72)),
                   shares=str(bits(d0, 112, 72)), payout=str(bits(d0, 184, 72)),
                   penalty=str(bits(d1, 0, 72)))
        if kind == 'stakeEnd':
            row.update(served=bits(d1, 72, 16), prevUnlocked=bits(d1, 88, 1))
        else:
            row['sender'] = '0x' + t[3][-40:]
    else:  # shareRateChange
        row.update(id=int(t[1], 16), ts=bits(d0, 0, 40), shareRate=bits(d0, 40, 40))
    return row

# ---- the HSI NFT's own transfers (2026-09-12, BossA: "a" — add the scan and a third bar) ----
# The manager IS the ERC-721, so a tokenized HSI moving between wallets is a
# plain Transfer on the same address.  Mints and burns carry the zero address
# and are the tokenize / detokenize already collected, so the publish step
# drops them and keeps wallet-to-wallet only.  The event has no timestamp of
# its own — unlike every other HSIM event — so the row carries its block and
# the publish step dates it from the block/timestamp pairs the other files
# already hold.
TOPICS_NFT = {
    '0xddf252ad1be2c89b69c2b068fc378daa952ba7f163c4a11628f55a4df523b3ef': 'nftTransfer',   # Transfer(address,address,uint256)
}

def decode_nft(kind, log):
    t = log['topics']
    return {'b': int(log['blockNumber'], 16), 'li': int(log['logIndex'], 16),
            'from': '0x' + t[1][-40:], 'to': '0x' + t[2][-40:], 'tid': int(t[3], 16)}

MAX_FAILS = 20   # consecutive failures with no progress before the run ends non-zero

def scan_era(era, url, frm, to, max_span, address=HEX, topics=TOPICS, decoder=decode, tag=''):
    start0 = frm
    ck_path = os.path.join(OUT, f'checkpoint-{era}{tag}.json')
    if os.path.exists(ck_path):
        frm = json.load(open(ck_path))['next']
        print(f'[{era}{tag}] resuming at {frm:,}', flush=True)
    files = {k: open(os.path.join(OUT, f'{era}-{k}.jsonl'), 'a') for k in topics.values()}
    span, req_n, ev_n, t_start = max_span, 0, 0, time.time()
    # 2026-09-13: a bad door must cost minutes, not the job's whole 90.  The 09-12 robot run
    # looped here in silence for 90 minutes — the error branch printed nothing and never gave
    # up.  Now every error is logged and MAX_FAILS in a row with no progress ends the run
    # non-zero.  And Dwellir's "exceeds the N-block limit" answer names its cap, so the span
    # jumps straight to it and the grow-back stops overshooting it: before, every success grew
    # the span to 750, the next call was refused, and a second was slept — half of every scan.
    fails, last_msg = 0, None
    while frm <= to:
        upto = min(frm + span - 1, to)
        try:
            r = rpc(url, 'eth_getLogs', [{'address': address, 'topics': [list(topics)],
                                          'fromBlock': hex(frm), 'toBlock': hex(upto)}])
        except Exception as e:
            fails += 1
            span = max(500, span // 2)
            print(f'[{era}{tag}] {frm:,}: {type(e).__name__}: {str(e)[:120]} -> span {span:,} ({fails}/{MAX_FAILS})', flush=True)
            if fails >= MAX_FAILS:
                sys.exit(f'[{era}{tag}] giving up at {frm:,}: {MAX_FAILS} failures in a row with no progress')
            time.sleep(3)
            continue
        if 'error' in r:
            msg = r['error'].get('message', '')
            m = re.search(r'retry with the range (\d+)-(\d+)', msg)
            cap = re.search(r'at most (\d+) blocks', msg)
            if m:
                span = int(m.group(2)) - int(m.group(1)) + 1
            elif cap and int(cap.group(1)) < span:
                span = max_span = int(cap.group(1))   # the plan's cap: never ask for more again
            else:
                fails += 1
                span = max(500, span // 2)
                if msg != last_msg or fails % 5 == 0:
                    print(f'[{era}{tag}] {frm:,}: rpc error: {msg[:160]} -> span {span:,} ({fails}/{MAX_FAILS})', flush=True)
                last_msg = msg
                if fails >= MAX_FAILS:
                    sys.exit(f'[{era}{tag}] giving up at {frm:,}: {MAX_FAILS} failures in a row with no progress — last error: {msg[:200]}')
                time.sleep(1)
            continue
        fails, last_msg = 0, None
        req_n += 1
        for log in r['result']:
            kind = topics[log['topics'][0]]
            files[kind].write(json.dumps(decoder(kind, log), separators=(',', ':')) + '\n')
        ev_n += len(r['result'])
        frm = upto + 1
        json.dump({'next': frm}, open(ck_path, 'w'))
        if req_n % 20 == 0:
            for f in files.values(): f.flush()
            pct = 100 * (1 - (to - frm) / max(1, to - start0))
            print(f'[{era}{tag}] {frm:,}/{to:,} ({pct:.1f}%) — {ev_n:,} events, {req_n} reqs, {time.time()-t_start:.0f}s', flush=True)
        # grow back toward the cap after a dense patch
        span = min(max_span, int(span * 1.5))
    for f in files.values(): f.close()
    print(f'[{era}{tag}] DONE: {ev_n:,} events in {req_n} requests, {time.time()-t_start:.0f}s', flush=True)

if __name__ == '__main__':
    tip = int(rpc(PLS_URL, 'eth_blockNumber', [])['result'], 16)
    print(f'PLS tip fixed at {tip:,}', flush=True)
    if '--hsim' not in sys.argv:
        scan_era('eth', ETH_URL, ETH_SCAN_FROM, FORK_FIRST_PLS - 1, 100_000)
        scan_era('pls', PLS_URL, FORK_FIRST_PLS, tip, 100_000)
    # the HSI manager's events, own checkpoints (checkpoint-<era>-hsim.json):
    # the Ethereum era from the manager's deploy, the PulseChain era from the fork
    scan_era('eth', ETH_URL, HSIM_SCAN_FROM, FORK_FIRST_PLS - 1, 100_000, address=HSIM, topics=TOPICS_HSIM, decoder=decode_hsim, tag='-hsim')
    scan_era('pls', PLS_URL, FORK_FIRST_PLS, tip, 100_000, address=HSIM, topics=TOPICS_HSIM, decoder=decode_hsim, tag='-hsim')
    # the NFT's own transfers, own checkpoints (checkpoint-<era>-nft.json)
    scan_era('eth', ETH_URL, HSIM_SCAN_FROM, FORK_FIRST_PLS - 1, 100_000, address=HSIM, topics=TOPICS_NFT, decoder=decode_nft, tag='-nft')
    scan_era('pls', PLS_URL, FORK_FIRST_PLS, tip, 100_000, address=HSIM, topics=TOPICS_NFT, decoder=decode_nft, tag='-nft')
    print('backfill complete', flush=True)
