"""Parse `Affixes list.rb` (RPG Maker VX Ace data) en JSON pour le simulateur d'infusion.

Le fichier Ruby contient deux blocs avec les mêmes indices (5..60) :
1. Encyclopédie : `AFFIXES[N][:enc_info] = {:name, :item, :category, :chance, ...}`
2. Effets      : `N => {:name (fragment), :color, :rarity, :statP, :SstatP, :features}`

Sortie : `public/data/affixes_csi_forever.json`
"""
import json
import re
import sys
from pathlib import Path

ROOT = Path(r"c:\wamp64\www\csi-world")
SRC  = ROOT / "Affixes list.rb"
DST  = ROOT / "public" / "data" / "affixes_csi_forever.json"

STAT_KEYS      = ["atkP", "defP", "matP", "mdfP", "agiP", "lukP", "hpP", "mpP"]
TIER_STAT_KEYS = ["SatkP", "SdefP", "SmatP", "SmdfP", "SagiP", "SlukP", "ShpP", "SmpP"]


def parse_enc_info(text: str) -> dict[int, dict]:
    """Bloc 1 : AFFIXES[N][:enc_info] = { ... }"""
    pattern = re.compile(r'AFFIXES\[(\d+)\]\[:enc_info\]\s*=\s*\{([^}]+)\}', re.DOTALL)
    rows = {}
    for m in pattern.finditer(text):
        idx  = int(m.group(1))
        body = m.group(2)
        rec  = {}
        sm = re.search(r':name\s*=>\s*"([^"]+)"', body)
        if sm: rec['displayName'] = sm.group(1)
        sm = re.search(r':item\s*=>\s*(\d+)', body)
        if sm: rec['itemRvId'] = int(sm.group(1))
        sm = re.search(r':category\s*=>\s*\[([^\]]+)\]', body)
        if sm: rec['categories'] = [int(x) for x in sm.group(1).split(',')]
        sm = re.search(r':chance\s*=>\s*(\d+)', body)
        if sm: rec['chance'] = int(sm.group(1))
        rows[idx] = rec
    return rows


def parse_effects(text: str) -> dict[int, dict]:
    """Bloc 2 : N => { ... }, après le marqueur `#####Effets#####`."""
    start = text.find('#####Effets#####')
    if start == -1:
        raise RuntimeError('Marqueur "#####Effets#####" introuvable')
    body = text[start:]

    # Match `N => { ... }`. Le corps ne contient jamais de `{}` (les features
    # utilisent `[]`), donc `[^{}]*` est sûr et évite les pièges de lookahead
    # autour des blocs de commentaires entre affixes.
    pattern = re.compile(
        r'^\s*(\d+)\s*=>\s*\{([^{}]*)\}',
        re.DOTALL | re.MULTILINE
    )
    rows = {}
    for m in pattern.finditer(body):
        rows[int(m.group(1))] = parse_effect_body(m.group(2))
    return rows


def parse_effect_body(body: str) -> dict:
    rec = {'stats': {}, 'tierStats': {}, 'features': []}

    sm = re.search(r':name\s*=>\s*"([^"]*)"', body)
    if sm: rec['fragment'] = sm.group(1)

    sm = re.search(r':color\s*=>\s*Color\.new\((\d+),\s*(\d+),\s*(\d+)\)', body)
    if sm:
        r, g, b = (int(sm.group(i)) for i in (1, 2, 3))
        rec['color'] = f'#{r:02x}{g:02x}{b:02x}'

    sm = re.search(r':rarity\s*=>\s*(\d+)', body)
    if sm: rec['rarity'] = int(sm.group(1))

    for key in STAT_KEYS:
        sm = re.search(rf':{key}\s*=>\s*(-?\d+)', body)
        if sm: rec['stats'][key] = int(sm.group(1))

    for key in TIER_STAT_KEYS:
        sm = re.search(rf':{key}\s*=>\s*(-?\d+)', body)
        if sm: rec['tierStats'][key] = int(sm.group(1))

    # Features : on extrait tous les triples [N, M, V] directement du body.
    # Le bloc effets ne contient pas d'autres arrays à 3 éléments, donc safe.
    for fm in re.finditer(r'\[\s*(-?\d+)\s*,\s*(-?\d+)\s*,\s*(-?[\d.]+)\s*\]', body):
        code    = int(fm.group(1))
        data_id = int(fm.group(2))
        value   = float(fm.group(3))
        if value == int(value):
            value = int(value)
        rec['features'].append([code, data_id, value])

    return rec


def determine_kind(fragment: str | None, idx: int) -> str:
    """Préfixe = espace en fin (« Grand(e) »), suffixe = espace en début (« foudroyant(e) »).
    Fallback sur l'index : 5-8 = préfixes, 9-60 = suffixes (couvre les cas avec espace
    des deux côtés comme ' allégé(e) ' ou fragment vide AFFIXES[5])."""
    if fragment:
        starts = fragment[0].isspace()
        ends   = fragment[-1].isspace()
        if starts and not ends:  return 'suffix'
        if ends and not starts:  return 'prefix'
    # Fragment vide ou ambigu : on retombe sur la range
    return 'prefix' if 5 <= idx <= 8 else 'suffix'


def main():
    text = SRC.read_text(encoding='utf-8')
    enc  = parse_enc_info(text)
    eff  = parse_effects(text)

    all_ids = sorted(set(enc.keys()) | set(eff.keys()))
    affixes = []
    for idx in all_ids:
        e1 = enc.get(idx, {})
        e2 = eff.get(idx, {})
        fragment = e2.get('fragment', '')
        affixes.append({
            'id':         idx,
            'name':       e1.get('displayName', ''),
            'fragment':   fragment,
            'kind':       determine_kind(fragment, idx),
            'categories': e1.get('categories', []),
            'itemRvId':   e1.get('itemRvId'),
            'rarity':     e2.get('rarity', 1),
            'color':      e2.get('color'),
            'chance':     e1.get('chance'),
            'stats':      e2.get('stats', {}),
            'tierStats':  e2.get('tierStats', {}),
            'features':   e2.get('features', []),
        })

    DST.parent.mkdir(parents=True, exist_ok=True)
    DST.write_text(json.dumps(affixes, ensure_ascii=False, indent=2), encoding='utf-8')

    # Récap
    print(f'OK : {len(affixes)} affixes → {DST.relative_to(ROOT)}')
    kinds = {}
    for a in affixes:
        kinds[a['kind']] = kinds.get(a['kind'], 0) + 1
    print(f'Répartition : {kinds}')
    missing_eff = [a['id'] for a in affixes if a['id'] not in eff]
    missing_enc = [a['id'] for a in affixes if a['id'] not in enc]
    if missing_eff: print(f'⚠ sans effets : {missing_eff}')
    if missing_enc: print(f'⚠ sans enc_info : {missing_enc}')


if __name__ == '__main__':
    sys.stdout.reconfigure(encoding='utf-8')
    main()
