#!/usr/bin/env python3
"""SkillFishOS Tuner - privileged DAEMON (root via pkexec, started once at app launch).
Reads one JSON command per line on stdin, writes one JSON reply per line on stdout.
This way the user authenticates ONCE at startup, not per-action."""
import sys, os, json, subprocess, re, glob, time

# Gli strumenti di overclock vanno cercati in /opt, non in /root.
#
# Avevano il percorso /root/... scritto dentro, e funzionavano solo sulla
# scheda di sviluppo: eggs esclude /root dall'immagine (root/* e root/.* sono
# fra le sue esclusioni predefinite, a meno di --includeRootHome, che noi non
# passiamo). Verificato montando la 26.06.3-rc2: dentro lo squashfs /root e'
# vuota. Quindi su OGNI sistema installato bc250_apply.py e bc250memcfg non
# esistevano, e il pulsante "applica" del Tuner falliva in silenzio.
#
# bc250-smu-oc.service era gia' stato spostato su /opt e infatti l'overclock
# all'avvio funzionava: era rimasto indietro solo il percorso di qui.
#
# /root resta come ricaduta per le macchine gia' installate che ce l'hanno.
def _first_dir(*paths):
    for p in paths:
        if os.path.isdir(p):
            return p
    return paths[0]

def _first_file(*paths):
    for p in paths:
        if os.path.exists(p):
            return p
    return paths[0]

OC_DIR  = _first_dir("/opt/bc250_smu_oc", "/root/bc250_smu_oc")
OC_CONF = "/etc/bc250-smu-oc.conf"
GOV_CONF= "/etc/cyan-skillfish-governor/config.toml"
# skillfish-memcfg e' NOSTRO (GPL-3.0) e sta in un pacchetto, mentre lo
# strumento che stava in /opt era di terzi, senza licenza dichiarata e
# quindi non ridistribuibile: non poteva stare nella ISO e su una macchina
# installata da apt non c'era proprio. I due percorsi vecchi restano come
# ripiego per le schede aggiornate da una versione precedente.
MEMCFG  = _first_file("/usr/local/bin/skillfish-memcfg",
                      "/opt/bc250_memcfg/bc250memcfg",
                      "/root/bc250_memcfg/bc250memcfg")
def find_vkpeak():
    """Where vkpeak is, looked up on every call.

    skillfish-vkpeak installs it in /usr/lib/skillfish/vkpeak (issue #91). The
    two bench folders are where the development boards and the images cloned
    from them had it; no package ever put it there. Looked up each time rather
    than once at start, so installing the package while this helper is running
    is enough.
    """
    for pattern in ("/usr/lib/skillfish/vkpeak/vkpeak", "/opt/bench/vkpeak*/vkpeak", "/root/bench/vkpeak*/vkpeak"):
        found = sorted(glob.glob(pattern))
        if found and os.access(found[0], os.X_OK):
            return found[0]
    return None

def _rd(path, mode="r"):
    """Read a whole file with a context manager (no leaked fd)."""
    with open(path, mode) as f:
        return f.read()

NL = chr(10)  # a capo, senza scriverlo come sequenza


def _wr(path, data):
    """Write a whole file with a context manager (no leaked fd)."""
    with open(path, "w") as f:
        f.write(data)

def sh(cmd, t=120):
    try: return subprocess.run(cmd, shell=True, capture_output=True, text=True, timeout=t)
    except Exception as e:
        class R: returncode=1; stdout=""; stderr=str(e)
        return R()

def nct_dir():
    """hwmon of the NCT6686 SuperIO that can actually drive the fan.

    Two drivers bind to this chip on the BC-250: the mainline nct6683 (forced) and
    the out-of-tree nct6687. Both register an hwmon named "nct6686", but only one
    exposes pwm2_enable — and which of the two gets the lower hwmon number varies
    between boots. Taking the first match silently wrote fan settings into the
    read-only node, so the fan never moved. Prefer the controllable one.
    """
    cands = []
    for h in sorted(glob.glob("/sys/class/hwmon/hwmon*")):
        try:
            if _rd(h+"/name").strip() == "nct6686": cands.append(h)
        except Exception: pass  # best-effort: skip hwmon nodes we can't read
    for h in cands:
        if os.path.exists(h+"/pwm2_enable"): return h
    return cands[0] if cands else None

def temp(name):
    for h in glob.glob("/sys/class/hwmon/hwmon*"):
        try:
            if _rd(h+"/name").strip()==name:
                return int(_rd(h+"/temp1_input"))//1000
        except Exception: pass  # best-effort: sensor may be absent/unreadable
    return 0

def cpu_min_freq():
    fs=[float(x) for x in re.findall(r'cpu MHz\s*:\s*([\d.]+)', _rd("/proc/cpuinfo"))]
    return int(min(fs)) if fs else 0

def active_cu():
    # ⚠️ THE FILE IS /run/skillfish/cu_active AND IT SAYS "N/40", NOT "N".
    # This used to read /run/skillfish-cu, a path nobody writes, and then
    # demanded isdigit(), which "40/40" is not. Both wrong, so the panel showed
    # "Active CUs: 0" next to a grid reading 40/40. Reported by Aviv Becker,
    # 23/08/2026, from a live BC-250.
    for percorso in ("/run/skillfish/cu_active", "/run/skillfish-cu"):
        try:
            v=_rd(percorso).strip()
            if not v: continue
            n=v.split("/")[0].strip()
            if n.isdigit(): return int(n)
        except Exception: pass  # best-effort: runtime CU file may not exist yet
    m=re.search(r'active_cu_number\s+(\d+)', sh("dmesg | grep active_cu_number | tail -1").stdout)
    return int(m.group(1)) if m else 0

# ---------- read current ----------
def get():
    out={}
    cpu={"frequency":3700,"scale":0,"max_temperature":85}
    try:
        for line in _rd(OC_CONF).splitlines():
            m=re.match(r'(\w+)\s*=\s*(-?\d+)',line.strip())
            if m: cpu[m.group(1)]=int(m.group(2))
    except Exception: pass  # best-effort: OC conf may be missing -> keep defaults
    out["cpu"]=cpu
    gpu={"min_mhz":350,"min_mv":700,"max_mhz":2230,"max_mv":1000}
    try:
        pts=re.findall(r'frequency\s*=\s*(\d+)\s*\n\s*voltage\s*=\s*(\d+)', _rd(GOV_CONF))
        if len(pts)>=2:
            gpu["min_mhz"],gpu["min_mv"]=int(pts[0][0]),int(pts[0][1])
            gpu["max_mhz"],gpu["max_mv"]=int(pts[-1][0]),int(pts[-1][1])
    except Exception: pass  # best-effort: governor conf may be missing -> keep defaults
    gpu["gov_mode"]=current_gov_mode()
    out["gpu"]=gpu
    vram={"uma_mb":0}
    if os.path.exists(MEMCFG):
        m=re.search(r'UMA_SIZE=(\d+)', sh(MEMCFG + ' get').stdout)
        if m: vram["uma_mb"]=int(m.group(1))
    out["vram"]=vram
    fan={"mode":"auto","pct":50,"rpm":0}
    nd=nct_dir()
    if nd:
        try:
            en=_rd(nd+"/pwm2_enable").strip(); pwm=int(_rd(nd+"/pwm2")); rpm=int(_rd(nd+"/fan2_input"))
            fan={"mode":("manual" if en=="1" else "auto"),"pct":pwm*100//255,"rpm":rpm}
        except Exception: pass  # best-effort: fan node may be unreadable
    out["fan"]=fan
    cu={"active":active_cu(),"max":40,"floor":7,"rows":{"0.0":7,"0.1":7,"1.0":7,"1.1":7},"live":False}
    try:
        j=json.loads(sh("/usr/local/bin/skillfish-cu get").stdout)
        cu["active"]=j.get("active_cu",cu["active"]); cu["rows"]=j.get("rows",cu["rows"])
        cu["floor"]=j.get("floor",7); cu["live"]=True
    except Exception: pass  # best-effort: skillfish-cu may be unavailable
    out["cu"]=cu
    return out

# ---------- apply ----------
def apply_cpu(mhz,scale,tmp):
    _wr(OC_CONF,"[overclock]\nfrequency = %d\nscale = %d\nmax_temperature = %d\n"%(mhz,scale,tmp))
    sh("systemctl stop cyan-skillfish-governor")
    r=sh("python3 %s/bc250_apply.py --apply %s"%(OC_DIR,OC_CONF))
    sh("systemctl start cyan-skillfish-governor")
    return r.returncode==0

def persist_cpu():
    # the unit comes with skillfish-smu-oc: --install would replace it with upstream's, unguarded
    sh("systemctl daemon-reload; systemctl enable bc250-smu-oc.service")

def _gpu_curve(minmhz,minmv,maxmhz,maxmv):
    # Build a SMOOTH multi-point voltage curve (gentle clock/voltage transitions).
    # The BC-250 SMU can hard-hang on abrupt jumps, so insert validated mid-points
    # (1500/900, 2000/1000) between idle and the requested max instead of a 2-point line.
    # Safety clamp: >2200 MHz at <=1000 mV is UNDERVOLTED and hard-freezes the
    # machine (reproduced + community data: 2230 needs 1000-1060 mV). Allow >2200
    # only when the caller raises the voltage accordingly.
    maxmhz, maxmv = int(maxmhz), int(maxmv)
    if maxmhz > 2200 and maxmv <= 1000:
        maxmhz = 2200
    pts=[(int(minmhz),int(minmv))]
    for f,v in ((1500,900),(2000,1000)):
        if minmhz < f < maxmhz: pts.append((f,v))
    pts.append((int(maxmhz),int(maxmv)))
    # de-dup by frequency, keep ascending
    seen=set(); out=[]
    for f,v in pts:
        if f in seen: continue
        seen.add(f); out.append((f,v))
    return out

def _log(msg):
    """Una riga nel registro di sistema. Si legge con
       journalctl -t skillfish-tuner-helper
    Non esisteva: il helper faceva cose che cambiano l'hardware senza lasciare
    traccia da nessuna parte."""
    try:
        import syslog
        syslog.openlog("skillfish-tuner-helper")
        syslog.syslog(msg)
        syslog.closelog()
    except Exception:
        pass


def _chi_mi_ha_chiamato():
    """La riga di comando del processo padre, per il registro.

    Il helper viene lanciato e muore subito: da fuori si vede solo che
    «qualcosa» ha riscritto la curva. Il padre invece resta, ed e' lui il
    responsabile."""
    try:
        with open("/proc/%d/stat" % os.getppid()) as f:
            ppid = int(f.read().split(") ", 1)[1].split()[1])
        with open("/proc/%d/cmdline" % ppid, "rb") as f:
            return f.read().replace(b"\0", b" ").decode("utf-8", "replace").strip()[:160]
    except Exception:
        return "?"


def apply_gpu(minmhz,minmv,maxmhz,maxmv):
    # ⚠️ Ogni riscrittura della curva lascia traccia. Riscriverla e' una delle
    # poche cose che possono inchiodare la GPU a meta' frequenza, o farle
    # calcolare sbagliato, senza che nessuno se ne accorga: se succede, si deve
    # poter risalire a chi l'ha chiesto e con quali valori.
    try:
        _log("curva GPU riscritta: %s/%s -> %s/%s mV, chiesto da: %s"
             % (minmhz, minmv, maxmhz, maxmv, _chi_mi_ha_chiamato()))
    except Exception:
        pass
    txt=_rd(GOV_CONF)
    txt=re.sub(r'(\[\[safe-points\]\]\s*\nfrequency[^\n]*\nvoltage[^\n]*\n?)+','',txt)
    txt=txt.rstrip()+"\n"
    for f,v in _gpu_curve(minmhz,minmv,maxmhz,maxmv):
        txt+="[[safe-points]]\nfrequency = %d\nvoltage = %d\n"%(f,v)
    _wr(GOV_CONF,txt)
    # gentle reload (stop, settle, start) — avoids SMU hard-hang on abrupt transition
    sh("systemctl stop cyan-skillfish-governor"); sh("sleep 2")
    return sh("systemctl start cyan-skillfish-governor").returncode==0

# ---------- GPU governor mode (balanced / performance) ----------
# Balanced = upstream load-target (cooler, only clocks up as needed).
# Performance = low load-target band + snappier ramp -> holds the top safe-point
# under any real gaming load (best FPS in GPU-bound titles); still idles to 350.
GOV_BAL  = {"sample":2000,"adjust":20000,"normal":1,"burst":200,"bsamples":48,"fadj":100,"upper":"0.95","lower":"0.7"}
GOV_PERF = {"sample":1000,"adjust":10000,"normal":6,"burst":250,"bsamples":16,"fadj":50, "upper":"0.20","lower":"0.08"}

def _gov_safepoints():
    try:
        pts=re.findall(r'frequency\s*=\s*(\d+)[^\n]*\n\s*voltage\s*=\s*(\d+)', _rd(GOV_CONF))
        if len(pts)>=2:
            d={}
            for f,v in pts: d[int(f)]=int(v)   # de-dup by frequency
            return sorted(d.items())           # ascending
    except Exception: pass  # best-effort: keep safe defaults
    return [(350,700),(1500,900),(2000,1000),(2200,1000)]

def current_gov_mode():
    try:
        m=re.search(r'\[load-target\][^\[]*?upper\s*=\s*([0-9.]+)', _rd(GOV_CONF), re.S)
        if m and float(m.group(1))<=0.5: return "performance"
    except Exception: pass  # best-effort
    return "balanced"

def gov_mode(mode):
    p = GOV_PERF if mode=="performance" else GOV_BAL
    sp = _gov_safepoints()  # preserve the user's max-freq/voltage safe-points
    txt = ("# SkillFishOS Tuner - cyan-skillfish-governor (mode: %s). Managed via the Tuner.\n"
           "[timing.intervals]\nsample = %d\nadjust = %d\nfinetune = 1000000000\n\n"
           "[timing.ramp-rates]\nnormal = %d\nburst = %d\n\n"
           "[timing]\nburst-samples = %d\n\n"
           "[frequency-thresholds]\nadjust = %d\nfinetune = 10\n\n"
           "[load-target]\nupper = %s\nlower = %s\n"
          ) % (mode, p["sample"], p["adjust"], p["normal"], p["burst"], p["bsamples"], p["fadj"], p["upper"], p["lower"])
    for f,v in sp:
        txt += "\n[[safe-points]]\nfrequency = %d\nvoltage = %d\n" % (f,v)
    _wr(GOV_CONF, txt)
    # Gentle reload: the BC-250 SMU can hard-hang on an abrupt clock transition,
    # so stop the governor, let the GPU settle to idle, then start with the new config.
    sh("systemctl stop cyan-skillfish-governor")
    sh("sleep 2")
    return sh("systemctl start cyan-skillfish-governor").returncode==0

def fan_set(mode,pct):
    nd=nct_dir()
    if not nd: return False,0
    if mode=="manual":
        _wr(nd+"/pwm2_enable","1")
        _wr(nd+"/pwm2",str(max(0,min(255,int(pct)*255//100))))
    else:
        _wr(nd+"/pwm2_enable","2")
    time.sleep(1)
    try: return True,int(_rd(nd+"/fan2_input"))
    except Exception: return True,0

def set_vram(mb):
    """Cambia la memoria riservata alla GPU. Ha effetto al prossimo avvio.

    ⚠️ Ogni strumento ha la SUA sintassi giusta: il nostro skillfish-memcfg
    vuole `set 4096`, quello di terzi vuole `UMA_SIZE 4096`. Chiamare il
    secondo con la sintassi del primo non da' errore: non riconosce gli
    argomenti, non scrive niente ed esce con 0. Cosi' rispondevamo "fatto" su
    un lavoro mai svolto.
    """
    if not os.path.exists(MEMCFG): return False
    mb = max(256, int(mb))
    nostro = MEMCFG.endswith("skillfish-memcfg")
    cmd = "%s set %d" % (MEMCFG, mb) if nostro else "%s UMA_SIZE %d" % (MEMCFG, mb)
    if sh(cmd).returncode != 0:
        return False
    # E si rilegge: il codice di uscita dice che il programma non e' morto,
    # non che il valore sia cambiato.
    letto = sh("%s %s" % (MEMCFG, "get" if nostro else "")).stdout or ""
    import re as _re
    m = _re.search(r"UMA_SIZE\s*[=:]?\s*(\d+)", letto)
    return bool(m) and int(m.group(1)) == mb

def cu_test():
    """Health-test each extra CU pair (WGP3-4 on every SE/SH) for the silicon
    lottery: enable the 24-CU floor + one extra WGP at a time, stress it with
    vkpeak, and flag GPU faults/hangs or a WGP that adds no performance.
    Restores the previous CU config at the end. ~2-3 min."""
    import time
    VKPEAK = find_vkpeak()
    if not VKPEAK: return {"ok":False,"err":"vkpeak is not installed: sudo apt install skillfish-vkpeak"}
    vdir=os.path.dirname(VKPEAK)
    def vk():
        # stdbuf -> line-buffered so the early fp32-scalar line survives the timeout kill
        r=sh("cd %s && stdbuf -oL -eL timeout 22 ./vkpeak"%vdir, t=35)
        m=re.search(r'fp32-scalar\s*=\s*([\d.]+)', r.stdout)
        return (float(m.group(1)) if m else 0.0), r.returncode
    def alive():
        return "BC-250" in sh("timeout 12 vulkaninfo --summary 2>/dev/null | grep -m1 deviceName", t=20).stdout
    def errs():
        s=sh("dmesg --since '16 seconds ago' 2>/dev/null | grep -ciE 'amdgpu.*(fault|timeout|reset|hang|recover|failed)'").stdout.strip()
        return int(s) if s.isdigit() else 0
    try: cur=json.loads(sh("/usr/local/bin/skillfish-cu get").stdout).get("rows",{})
    except Exception: cur={}
    order=["0.0","0.1","1.0","1.1"]
    res=[]
    sh("/usr/local/bin/skillfish-cu set-rows 7 7 7 7"); time.sleep(1)
    base,_=vk()
    for rk in order:
        for wgp in (3,4):
            rows={k:7 for k in order}; rows[rk]=7|(1<<wgp)
            sh("/usr/local/bin/skillfish-cu set-rows %d %d %d %d"%(rows["0.0"],rows["0.1"],rows["1.0"],rows["1.1"]))
            time.sleep(1)
            g,rc=vk(); e=errs(); al=alive()
            # defect = GPU fault/hang/no-response under load (the silicon-lottery symptom).
            # per-WGP throughput delta is below vkpeak noise, so verdict is stability-based.
            if not al or e>0: verdict="FAIL"
            elif g<=0:        verdict="N/A"
            else:             verdict="OK"
            res.append({"row":rk,"wgp":wgp,"cu":"%d-%d"%(wgp*2,wgp*2+1),
                        "gflops":round(g),"errors":e,"verdict":verdict})
            time.sleep(1.5)
    # headline: full 40-CU sustained run (vkpeak to completion) + GPU error scan
    sh("/usr/local/bin/skillfish-cu max"); time.sleep(1)
    rf=sh("cd %s && ./vkpeak"%vdir, t=120)
    mf=re.search(r'fp32-scalar\s*=\s*([\d.]+)', rf.stdout)
    full=round(float(mf.group(1)) if mf else 0.0); full_err=errs()
    g0=lambda k: int(cur.get(k,31))
    sh("/usr/local/bin/skillfish-cu set-rows %d %d %d %d"%(g0("0.0"),g0("0.1"),g0("1.0"),g0("1.1")))
    bad=sum(1 for x in res if x["verdict"]=="FAIL")
    na=sum(1 for x in res if x["verdict"]=="N/A")
    return {"ok":True,"baseline":round(base),"results":res,"bad":bad,"na":na,
            "full40":full,"full40_err":full_err}

def cu_apply(rows):
    """Apply live per-row WGP masks (24..40 CU) via skillfish-cu. No reboot.

    ⚠️ L'ERRORE DI skillfish-cu VA RIPORTATO, NON BUTTATO.
    Prima qui si teneva solo il codice di uscita e si restituiva {"ok": False}
    senza "err": il Tuner cadeva sul suo ripiego e mostrava «apply failed», che
    non dice niente e non permette nemmeno di fare una domanda sensata. Un
    utente ci ha scritto proprio con quella frase e non avevamo modo di sapere
    cosa fosse andato storto.

    skillfish-cu l'errore ce l'ha e va bene com'e': «nessuna CU leggibile: umr
    non sta parlando con la GPU» oppure «chieste N CU, ottenute M». Si prende
    da stderr e si passa su.
    """
    try:
        # ⚠️ Due pezzi, e per un po' ne spedivamo uno solo: il programma umr e il
        # database dei registri che il programma legge. Senza il secondo umr
        # parte e fallisce con «Cannot open pci.did», che a un utente non dice
        # niente. Successo davvero, sulla seconda BC-250 installata da ISO.
        if not os.path.exists("/usr/local/bin/umr"):
            return {"ok": False, "err": "manca /usr/local/bin/umr, "
                    "senza il quale le CU non si possono instradare"}
        if not os.path.exists("/usr/local/share/umr/database/pci.did"):
            return {"ok": False, "err": "manca il database dei registri di umr "
                    "(/usr/local/share/umr/database): reinstalla skillfish-tuner"}
        r=[int(x)&0x1f for x in rows][:4]
        while len(r)<4: r.append(0x07)
        p=sh("/usr/local/bin/skillfish-cu set-rows %s"%(" ".join(str(x) for x in r)))
        j=json.loads(sh("/usr/local/bin/skillfish-cu get").stdout)
        if p.returncode!=0:
            motivo=(getattr(p,"stderr","") or "").strip() or (p.stdout or "").strip()
            if not motivo:
                motivo="skillfish-cu e' uscito con %d senza dire perche'"%p.returncode
            return {"ok":False,"err":motivo,"uscita":p.returncode,
                    "active":j.get("active_cu"),"rows":j.get("rows")}
        return {"ok":True,"active":j.get("active_cu"),"rows":j.get("rows")}
    except Exception as e:
        return {"ok":False,"err":str(e)}

def cpu_cores_get():
    """Live CPU-core map: one entry per physical core with its SMT siblings.

    Every logical CPU can be taken offline except cpu0 — it is the boot CPU and
    this kernel is built without BOOTPARAM_HOTPLUG_CPU0 — so core 0 can drop its
    second thread but never disappear entirely.
    """
    cores = {}
    for p in sorted(glob.glob("/sys/devices/system/cpu/cpu[0-9]*")):
        n = os.path.basename(p)[3:]
        if not n.isdigit(): continue
        n = int(n)
        try:
            core = int(_rd("%s/topology/core_id" % p).strip())
        except Exception:
            continue  # offline CPUs hide their topology; folded in below
        online = True if not os.path.exists(p + "/online") else _rd(p + "/online").strip() == "1"
        cores.setdefault(core, {"core": core, "cpus": [], "online": False, "removable": True})
        cores[core]["cpus"].append({"cpu": n, "online": online,
                                    "removable": os.path.exists(p + "/online")})
        if online: cores[core]["online"] = True
        if not os.path.exists(p + "/online"): cores[core]["removable"] = False
    # an offline CPU loses topology/core_id, so recover it from the sibling numbering
    for p in sorted(glob.glob("/sys/devices/system/cpu/cpu[0-9]*")):
        n = os.path.basename(p)[3:]
        if not n.isdigit(): continue
        n = int(n)
        if any(c["cpu"] == n for e in cores.values() for c in e["cpus"]): continue
        core = n // 2   # this APU pairs threads as (2k, 2k+1)
        cores.setdefault(core, {"core": core, "cpus": [], "online": False, "removable": True})
        cores[core]["cpus"].append({"cpu": n, "online": False, "removable": True})
    smt = None
    try: smt = _rd("/sys/devices/system/cpu/smt/control").strip()
    except Exception: pass
    out = [cores[k] for k in sorted(cores)]
    for e in out: e["cpus"].sort(key=lambda c: c["cpu"])
    return {"ok": True, "cores": out, "nproc": os.cpu_count(), "smt": smt}


def cpu_cores_set(states):
    """states: list of {core, online}. Toggles both SMT siblings of each core.

    Refuses to leave the machine with no cores and never touches cpu0, so a bad
    request can't strand the box — the worst case is one core left running.
    """
    want = {}
    for s in states or []:
        try: want[int(s["core"])] = bool(s["online"])
        except Exception: pass
    if not want: return {"ok": False, "err": "nessun core specificato"}
    cur = cpu_cores_get()["cores"]
    if not any(want.get(e["core"], e["online"]) for e in cur):
        return {"ok": False, "err": "almeno un core deve restare acceso"}
    for e in cur:
        tgt = want.get(e["core"])
        if tgt is None: continue
        for c in e["cpus"]:
            if not c["removable"]: continue          # cpu0
            _wr("/sys/devices/system/cpu/cpu%d/online" % c["cpu"], "1" if tgt else "0")
    time.sleep(1)
    return cpu_cores_get()


def cpu_smt_set(on):
    """Global SMT switch — halves or restores the thread count in one go."""
    p = "/sys/devices/system/cpu/smt/control"
    if not os.path.exists(p): return {"ok": False, "err": "SMT non controllabile su questo kernel"}
    _wr(p, "on" if on else "off")
    time.sleep(1)
    return cpu_cores_get()


def set_cu(unlock):
    """Gone. The compute units are routed live now, not by a boot parameter.

    This used to add or remove amdgpu.bc250_cc_write_mode=3 in /etc/default/grub
    and then ask for a reboot. Two separate reasons it had to go.

    It is obsolete. skillfish-cu.service writes the same WGP masks at boot, and
    "cu-apply" below changes them while the machine is running. Measured on the
    dev board on 2026-09-17: with the parameter removed and the board rebooted,
    still 40/40 CU and 10169 GFLOPS against 10166 with it.

    ⚠️ And it was broken on exactly the machines that matter. It appended the
    parameter with rstrip of a double quote followed by a literal double quote,
    which assumes the value is double-quoted. Calamares writes it single-quoted,
    so on an installed system the result was

        GRUB_CMDLINE_LINUX_DEFAULT='... 'amdgpu.bc250_cc_write_mode=3"

    with mismatched quotes, in the one file that decides whether the machine
    boots. Nothing in the tree calls this any more, so nobody has hit it, which
    is luck rather than design.

    The command is answered rather than dropped, so an old caller is told what
    to use instead of getting a bare "unknown command".
    """
    return {"ok": False, "reboot": False,
            "err": "le compute unit si cambiano a caldo: usa cu-apply, senza riavvio"}

# ---------- BENCHMARK / TEST ----------
def _govs_get():
    """Current governor per CPU. An offline CPU keeps its cpufreq node around but
    answers EBUSY, so read each one defensively — a core the user switched off in
    the Tuner must not blow up the benchmark."""
    out = {}
    for p in glob.glob("/sys/devices/system/cpu/cpu*/cpufreq/scaling_governor"):
        try: out[p] = _rd(p).strip()
        except OSError: pass   # offline core
    return out

def _govs_set(val):
    for p in glob.glob("/sys/devices/system/cpu/cpu*/cpufreq/scaling_governor"):
        try: _wr(p, val)
        except OSError: pass   # offline core

def bench_cpu(secs=60):
    """sysbench multi-thread per >=60s (temp realistica); campiona min-freq e temp max sotto carico.

    Since the ACPI P-state tables landed, this box finally has cpufreq — and with an
    idling governor a core legitimately drops to 800 MHz between sysbench phases.
    The min-freq sample then looked like an unstable overclock and every candidate
    was rejected. Pin the governor to performance for the measurement and put the
    user's choice back afterwards.
    """
    nth=os.cpu_count() or 6
    import threading
    prev_govs=_govs_get()
    if prev_govs: _govs_set("performance")
    samp={"minf":99999,"maxt":0}
    stop=threading.Event()
    def mon():
        # sysbench needs a moment to spin every thread up. Sampling from t=0 caught
        # the machine still idling and that single low reading became the "minimum",
        # failing even a known-good clock — 3500 was rejected at 1396 MHz while the
        # live frequency mid-bench was 3493. Let the load settle before believing it.
        deadline = time.monotonic() + 8
        while not stop.is_set() and time.monotonic() < deadline:
            time.sleep(0.5)
        while not stop.is_set():
            f=cpu_min_freq();  t=temp("k10temp")
            if f>0 and f<samp["minf"]: samp["minf"]=f
            if t>samp["maxt"]: samp["maxt"]=t
            time.sleep(2)
    th=threading.Thread(target=mon,daemon=True); th.start()
    r=sh("sysbench cpu --threads=%d --time=%d --cpu-max-prime=20000 run"%(nth,secs), t=secs+30)
    stop.set(); th.join(timeout=3)
    for p,g in prev_govs.items():
        try: _wr(p,g)          # restore the user's governor
        except OSError: pass   # core taken offline mid-bench
    m=re.search(r'events per second:\s*([\d.]+)', r.stdout)
    eps=float(m.group(1)) if m else 0
    minf=samp["minf"] if samp["minf"]<99999 else cpu_min_freq()
    return {"score":round(eps,1),"unit":"ev/s","min_mhz":minf,"temp":samp["maxt"] or temp("k10temp"),"ok":r.returncode==0 and eps>0}

def bench_gpu():
    VKPEAK = find_vkpeak()
    if not VKPEAK: return {"score":0,"unit":"GFLOPS","ok":False,"err":"vkpeak is not installed: sudo apt install skillfish-vkpeak"}
    r=sh("cd %s && ./vkpeak"%os.path.dirname(VKPEAK), t=150)
    # vkpeak prints lines like 'fp32-scalar  = 11329.xx GFLOPS'
    m=re.search(r'fp32-scalar\s*=\s*([\d.]+)', r.stdout)
    g=float(m.group(1)) if m else 0
    return {"score":round(g,0),"unit":"GFLOPS","temp":temp("amdgpu"),"ok":r.returncode==0 and g>0}

def _write_cpu_conf(mhz,scale,tmp):
    _wr(OC_CONF,"[overclock]\nfrequency = %d\nscale = %d\nmax_temperature = %d\n"%(mhz,scale,tmp))

def test_cpu(mhz,scale,tmp):
    prev=get()["cpu"]
    if not apply_cpu(mhz,scale,tmp):
        _write_cpu_conf(prev["frequency"],prev["scale"],prev["max_temperature"])
        return {"ok":False,"phase":"apply","err":"applicazione fallita"}
    # CRASH SAFETY: while the candidate is being benched, keep the LAST-KNOWN-GOOD
    # values on disk. A hard freeze mid-bench must not leave the unstable candidate
    # in the conf that bc250-smu-oc.service re-applies at boot (freeze loop).
    _write_cpu_conf(prev["frequency"],prev["scale"],prev["max_temperature"])
    b=bench_cpu()
    # stability: under load the min core freq should stay within 150MHz of target
    stable = b["ok"] and b["min_mhz"] >= (mhz-200)
    if not stable:
        apply_cpu(prev["frequency"],prev["scale"],prev["max_temperature"])  # rollback
        return {"ok":False,"applied":False,"bench":b,
                "err":"Instabile/throttle: %d MHz sotto carico (target %d). Ripristinato."%(b["min_mhz"],mhz)}
    _write_cpu_conf(mhz,scale,tmp)  # passed: persist the candidate
    return {"ok":True,"applied":True,"bench":b}

def suggest_uv(mhz):
    """Suggerisce l'undervolt (scale) ottimale per la frequenza data: applica, scende di
    scale finche' un breve stress resta stabile (min-freq non crolla). NON persiste:
    ripristina la config corrente alla fine. Ritorna il miglior scale trovato."""
    prev=get()["cpu"]
    best=0
    s=0
    while s>-20:  # limite di sicurezza
        if not apply_cpu(mhz, s, prev["max_temperature"]): break
        # crash safety: keep last-known-good on disk during the stress (see test_cpu)
        _write_cpu_conf(prev["frequency"],prev["scale"],prev["max_temperature"])
        b=bench_cpu(12)  # breve verifica
        if b["ok"] and b["min_mhz"] >= (mhz-200):
            best=s; s-=2
        else:
            break
    # ripristina lo stato iniziale
    apply_cpu(prev["frequency"],prev["scale"],prev["max_temperature"])
    return {"ok":True,"suggested_scale":best,"mhz":mhz}

def test_gpu(minmhz,minmv,maxmhz,maxmv):
    prev=get()["gpu"]
    if not apply_gpu(minmhz,minmv,maxmhz,maxmv):
        return {"ok":False,"phase":"apply","err":"applicazione fallita"}
    b=bench_gpu()
    if not b["ok"]:
        apply_gpu(prev["min_mhz"],prev["min_mv"],prev["max_mhz"],prev["max_mv"])
        return {"ok":False,"applied":False,"bench":b,"err":"Benchmark GPU fallito/instabile. Ripristinato."}
    return {"ok":True,"applied":True,"bench":b}

def thermal_guard(limit):
    # ⚠️ Qui dentro, fino al 27/08/2026, c'era l'INTERO script della guardia
    # termica incorporato in una stringa, e ogni chiamata lo riscriveva sopra
    # quello del pacchetto. Due copie della stessa logica in due posti: una
    # correzione fatta nel file vero spariva alla prima modifica del limite
    # dal Tuner. Peggio: la copia scritta da qui non aveva l'ExecCondition
    # che limita il servizio alla BC-250, quindi bastava toccare il limite
    # per togliere quella protezione su una macchina qualunque.
    # Adesso lo script sta in un posto solo, dentro il pacchetto, e qui si
    # scrive soltanto il limite che deve leggere.
    os.makedirs("/etc/skillfish", exist_ok=True)
    _wr("/etc/skillfish/thermal-guard.conf", "".join([
        "# Scritto da SkillFishOS Tuner. Temperatura oltre la quale la CPU",
        NL, "# viene rallentata di 100 MHz per volta.", NL,
        "LIMITE=%d" % int(limit), NL]))
    sh("systemctl enable --now skillfish-thermal-guard.service;"
       " systemctl restart skillfish-thermal-guard.service")
    return True

SEGNAPOSTO_CORE = "/etc/skillfish/core-unlock.abilitato"


def core_unlock_get():
    """Stato dello sblocco degli 8 core (6c/12t -> 8c/16t).

    Lo sblocco non e' piu' automatico dal 16/08/2026: scrive la maschera SMU e
    poi RIAVVIA, e una accensione a freddo la azzera, quindi il riavvio si
    ripete a ogni avvio da spenta. Sulla ISO rilasciata partiva da solo anche
    nella live e uccideva l'installazione a meta'. Adesso decide l'utente.
    """
    # niente shell: sched_getaffinity conta i thread davvero ONLINE, che e'
    # esattamente il numero che l'utente vede (12 bloccato, 16 sbloccato)
    try:
        n = len(os.sched_getaffinity(0))
    except (AttributeError, OSError):
        n = os.cpu_count() or 0
    return {"ok": True,
            "abilitato": os.path.exists(SEGNAPOSTO_CORE),
            "thread": n,
            "supportato": os.path.exists("/sys/bus/pci/devices/0000:00:00.0/config")}


def core_unlock_set(on):
    """Accende o spegne lo sblocco. Ha effetto dal prossimo avvio."""
    try:
        if on:
            os.makedirs("/etc/skillfish", exist_ok=True)
            _wr(SEGNAPOSTO_CORE,
                "# La presenza di questo file accende lo sblocco degli 8 core.\n"
                "# Lo gestisce SkillFishOS Tuner.\n")
        elif os.path.exists(SEGNAPOSTO_CORE):
            os.remove(SEGNAPOSTO_CORE)
    except OSError as e:
        return {"ok": False, "err": str(e)}
    d = core_unlock_get()
    d["riavvio"] = True
    return d


SERVIZIO_SCX = "skillfish-scx.service"
SCX_STATO = "/sys/kernel/sched_ext/state"
SCX_OPS = "/sys/kernel/sched_ext/root/ops"


def _testo(percorso):
    try:
        with open(percorso) as f:
            return f.read().strip()
    except OSError:
        return ""


def scx_get():
    """Stato dello schedulatore sched_ext (scx_lavd).

    ⚠️ Tre stati diversi, non uno: il servizio ABILITATO (torna al prossimo
    avvio), il servizio ATTIVO adesso, e lo schedulatore davvero CARICATO nel
    kernel. Se il kernel non ha sched_ext l'ultimo resta spento comunque, e
    l'interruttore va disattivato invece di far credere di aver acceso qualcosa.
    """
    supportato = (os.path.isdir("/sys/kernel/sched_ext")
                  and os.path.exists("/usr/local/bin/scx_lavd"))
    return {"ok": True,
            "supportato": supportato,
            "abilitato": sh("systemctl is-enabled " + SERVIZIO_SCX).returncode == 0,
            "attivo": sh("systemctl is-active " + SERVIZIO_SCX).returncode == 0,
            "caricato": _testo(SCX_STATO) == "enabled",
            "nome": _testo(SCX_OPS)}


def scx_set(on):
    """Accende o spegne lo schedulatore, adesso e ai prossimi avvii."""
    if not os.path.isdir("/sys/kernel/sched_ext"):
        return {"ok": False, "err": "questo kernel non ha sched_ext"}
    r = sh("systemctl %s %s" % ("enable --now" if on else "disable --now", SERVIZIO_SCX))
    if r.returncode != 0:
        return {"ok": False, "err": "systemctl ha rifiutato (%d)" % r.returncode}
    # ⚠️ systemctl torna appena il processo e' partito, ma lavd ci mette qualche
    # secondo a caricare il programma BPF e agganciarsi al kernel: misurato,
    # circa cinque. Leggendo subito si rispondeva «acceso ma non caricato», che
    # sembra un guasto e non lo e'. Si aspetta il kernel, con un tetto per non
    # restare appesi se qualcosa e' andato storto davvero.
    if on:
        for _ in range(20):
            if _testo(SCX_STATO) == "enabled":
                break
            time.sleep(0.5)
    return scx_get()


def handle(req):
    c=req.get("cmd")
    if c=="ping": return {"ok":True}
    if c=="get": return {"ok":True,"data":get()}
    if c=="apply-cpu": return {"ok":apply_cpu(req["mhz"],req["scale"],req["temp"])}
    if c=="persist-cpu": apply_cpu(req["mhz"],req["scale"],req["temp"]); persist_cpu(); return {"ok":True}
    if c=="apply-gpu": return {"ok":apply_gpu(req["minmhz"],req["minmv"],req["maxmhz"],req["maxmv"])}
    if c=="gov-mode": return {"ok":gov_mode(req["mode"]),"mode":current_gov_mode()}
    if c=="apply-fan":
        ok,rpm=fan_set(req["mode"],req.get("pct",50)); return {"ok":ok,"rpm":rpm}
    if c=="set-vram": return {"ok":set_vram(req["mb"]),"reboot":True}
    if c=="set-cu": return set_cu(req.get("unlock"))
    if c=="cu-apply": return cu_apply(req.get("rows",[]))
    if c=="cpu-cores": return cpu_cores_get()
    if c=="cpu-cores-set": return cpu_cores_set(req.get("cores",[]))
    if c=="cpu-smt": return cpu_smt_set(bool(req.get("on",True)))
    if c=="core-unlock": return core_unlock_get()
    if c=="core-unlock-set": return core_unlock_set(bool(req.get("on",False)))
    if c=="scx": return scx_get()
    if c=="scx-set": return scx_set(bool(req.get("on",False)))
    if c=="cu-test": return cu_test()
    if c=="thermal-guard": return {"ok":thermal_guard(req["limit"])}
    if c=="test-cpu": return test_cpu(req["mhz"],req["scale"],req["temp"])
    if c=="test-gpu": return test_gpu(req["minmhz"],req["minmv"],req["maxmhz"],req["maxmv"])
    if c=="suggest-uv": return suggest_uv(req["mhz"])
    return {"ok":False,"err":"comando sconosciuto"}

def main():
    # one-shot mode (for testing): arg 'get' prints once
    if len(sys.argv)>1 and sys.argv[1]=="get":
        print(json.dumps(get())); return
    # daemon mode: read JSON lines from stdin
    sys.stdout.write(json.dumps({"ok":True,"ready":True})+"\n"); sys.stdout.flush()
    for line in sys.stdin:
        line=line.strip()
        if not line: continue
        try: req=json.loads(line)
        except Exception: continue
        if req.get("cmd")=="quit": break
        try: rep=handle(req)
        except Exception as e: rep={"ok":False,"err":str(e)}
        sys.stdout.write(json.dumps(rep)+"\n"); sys.stdout.flush()

if __name__=="__main__": main()
