# -*- coding: utf-8 -*- """LEADBOARD — 신약 예측 도구를 분야별로 같은 잣대에 세우는 시험대. **이 서비스가 하는 일** · 카테고리·부문 카드를 서빙한다 · 테스트셋(구조만)을 내려준다 · 제출을 받아 비공개 원장에 적는다 · 워커가 굴려 놓은 순위표를 보여준다 **하지 않는 일: 채점.** 정답은 이 컨테이너에 없다. 컨테이너 이미지는 누구나 받을 수 있으므로, 여기에 라벨을 두면 그 순간 1조가 무너진다. 채점은 원장을 폴링하는 별도 워커가 로컬 정답 파일로 한다. 부문 카드에는 **채점을 검증하는 데 필요한 모든 공개 정보**가 들어 있다 — 분할 등급 · 정답 등급 · 잡음 바닥 · 기준선 성적 · 데이터 지문. 우리 점수를 믿어달라고 하지 않기 위해서다. """ import base64 import glob import hashlib import hmac import io import json import os import re import secrets import time import urllib.error import urllib.parse import urllib.request from fastapi import FastAPI, HTTPException, Request from fastapi.middleware.gzip import GZipMiddleware from fastapi.responses import (FileResponse, JSONResponse, RedirectResponse, Response) from pydantic import BaseModel HERE = os.path.dirname(os.path.abspath(__file__)) DATA = os.path.join(HERE, "data") SPEC = "v1.1" # 상금. 46부문 중 29부문이 비어 있고 외부 참가자가 1명뿐이라 붙인다. 순위표만 있고 # 왜 지금 해야 하는지가 없었다. # # 🔴 주최측 계정은 수상 대상에서 뺀다. 현재 1위 17개 중 16개가 주최측이라 그대로 두면 # 주최측이 자기 상금을 받는다. ODC 에서 기준물질을 표에 올리되 등수를 주지 않는 것과 # 같은 처리다 - 눈금 역할은 하되 겨루지는 않는다. HOST_ACCOUNTS = {"SeaWolf-AI", "FINAL-Bench", "VIDraft"} PRIZE = { "amount_usd": 1000, "winners": 1, "closes": "2027-12-31", # **1위 부문을 가장 많이 가진 한 사람.** 전 부문 제출 조건은 두지 않는다 # (회장님 2026-08-29 개정) - 문턱이 참가를 막으면 겨룰 사람 자체가 없다. # 부문마다 지표가 달라 점수를 합산할 수 없으므로 1위 부문의 개수로 센다. "rule": "most_first_places", "require_all_boards": False, } def is_host(user): return (user or "") in HOST_ACCOUNTS LEDGER = os.environ.get("LB_LEDGER_REPO", "FINAL-Bench/leadboard-submissions") LEDGER_API = "https://huggingface.co/api/datasets/%s" % LEDGER LEDGER_RAW = "https://huggingface.co/datasets/%s/resolve/main" % LEDGER HF_TOKEN = os.environ.get("HF_TOKEN", "") OAUTH_ID = os.environ.get("OAUTH_CLIENT_ID", "") OAUTH_SECRET = os.environ.get("OAUTH_CLIENT_SECRET", "") OAUTH_ISS = os.environ.get("OPENID_PROVIDER_URL", "https://huggingface.co") SPACE_HOST = os.environ.get("SPACE_HOST", "") COOKIE = "lb_session" IN_FRAME = bool(SPACE_HOST) COOKIE_KW = ({"samesite": "none", "secure": True} if IN_FRAME else {"samesite": "lax", "secure": False}) SESSION_KEY = os.environ.get("LB_SESSION_KEY") or secrets.token_hex(16) DAILY_CAP = int(os.environ.get("LB_DAILY_CAP", "5")) app = FastAPI(title="LEADBOARD") app.add_middleware(GZipMiddleware, minimum_size=1024) _C = {} # 하루 한도 카운터. 컨테이너 재기동이면 비는데, 그래도 폭주는 막는다. # 정확한 회계가 필요해지면 원장 쪽으로 옮긴다. _CAP = {} def cached(key, ttl, produce): hit = _C.get(key) if hit and time.time() - hit[0] < ttl: return hit[1] try: v = produce() except Exception: if hit: return hit[1] raise _C[key] = (time.time(), v) return v # ------------------------------------------------------------------ 부문 카드 def _sha(path): h = hashlib.sha256() with open(path, "rb") as f: for b in iter(lambda: f.read(65536), b""): h.update(b) return h.hexdigest()[:16] def _board_meta(): """부문 한 줄 설명. 이름과 표적 기호만으로는 무엇을 재는지 알 수 없다.""" p = os.path.join(HERE, "board_meta.json") try: return json.load(io.open(p, encoding="utf-8")).get("boards", {}) except Exception: return {} def _load_boards(): meta = _board_meta() out = {} for p in sorted(glob.glob(os.path.join(DATA, "*_card.json"))): try: d = json.load(io.open(p, encoding="utf-8")) except Exception: continue nf = d.get("noise_floor") or {} base = d.get("baselines") or {} # 축 개수의 정본은 시험셋 파일이고 카드에는 없다. 이 값을 여기서 안 실으면 # 제출 검증이 "화합물마다 숫자 하나"를 요구해 **축 부문에는 아무도 제출할 수 # 없다** (2026-09-08 참가자 제보). 화면 안내도 스칼라용 문구가 뜬다. # 카드에 넣지 않는 이유: dataset_sha 가 카드 해시라서, 카드를 고치면 # 데이터셋이 교체된 것처럼 기록된다. # mae 부문까지 시험셋을 열면 매 갱신마다 46개를 파싱하므로 특수 지표만 본다. if d.get("axes") is None and d.get("metric") not in (None, "mae"): tp = os.path.join(DATA, "%s_test.json" % str(d.get("board") or "").lower()) try: d["axes"] = json.load(io.open(tp, encoding="utf-8")).get("axes") except Exception: pass # 분류 전용 부문에는 MAE 기준선이 없다. 그럴 때는 AUROC 가 가장 높은 것을 최선으로 본다. # 없는 값을 0 으로 치면 그 부문이 "가장 정확한 곳"으로 표에 오른다. # 🔴 카드가 지정한 주지표를 덮어쓰지 않는다. 여기서 metric 을 무조건 다시 # 정하는 바람에, 프로파일 상관으로 채점되는 부문이 MAE 부문으로 바뀌어 있었다 # (그 부문에도 MAE 값이 부수적으로 실려 있어서 조건에 걸렸다). 화면·API· # 기준선 판정이 전부 그 값을 보고 있었으므로 한 줄이 세 곳을 동시에 틀리게 했다. card_metric = d.get("metric") with_mae = {k: v for k, v in base.items() if v.get("mae") is not None} if card_metric: best = (max(base, key=lambda k: base[k].get(card_metric) or 0) if card_metric != "mae" else (min(with_mae, key=lambda k: with_mae[k]["mae"]) if with_mae else None)) elif with_mae: best = min(with_mae, key=lambda k: with_mae[k]["mae"]) d["metric"] = "mae" elif base: best = max(base, key=lambda k: base[k].get("auroc") or 0) d["metric"] = "auroc" else: best = None d["metric"] = None d["dataset_sha"] = _sha(p) d["spec"] = SPEC d["best_baseline"] = best # 모델 오차가 실험 오차의 몇 배인가. 1 에 붙을수록 측정 한계다. if best and nf.get("sd_single") and d["metric"] == "mae": d["error_over_noise"] = round(base[best]["mae"] / nf["sd_single"], 2) # 학습 기준선이 상수 예측을 이기는가 - **그 부문의 주지표 위에서** 본다. # MAE 로만 보면 상관으로 채점되는 부문은 뜻 없는 비교가 되고, AUROC 전용 # 부문은 mae 가 None 이라 비교 자체가 없었다. mk, mhb = primary_metric(d) tr = (base.get("Morgan+LightGBM") or {}).get(mk) co = (base.get("상수 예측") or {}).get(mk) if tr is not None and co is not None: d["beats_constant"] = (tr > co) if mhb else (tr < co) # 격차를 함께 싣는다. "이겼다/졌다"만으로는 0.003 과 0.27 이 같아 보인다. d["beats_constant_margin"] = round((tr - co) if mhb else (co - tr), 4) if d.get("n_test"): d["near_pct"] = round(100.0 * d.get("near_threshold", 0) / d["n_test"], 1) m = meta.get(d["board"]) or {} d["blurb"] = m.get("ko", "") d["blurb_en"] = m.get("en", "") d["open"] = True out[d["board"]] = d return out def primary_metric(b): """이 부문의 주지표와 방향. (key, higher_better) 표현형은 프로파일 상관, 분류 전용은 AUROC, 나머지는 MAE 다. 판정을 여기 한 곳에 두는 이유는 채점기와 화면이 서로 다른 지표를 보고 있었기 때문이다 - Phenotype 는 상관으로 채점되는데 화면은 MAE 만 띄워서, 학습 기준선이 상수 예측을 0.003 차이로 간신히 이기는 것처럼 보였다(실제 상관으로는 0.092 대 0.159, 잡음 바닥의 7.6배). """ if (b or {}).get("metric") == "profile_corr": return "profile_corr", True bl = (b or {}).get("baselines") or {} if bl and all(v.get("mae") is None for v in bl.values()) and any(v.get("auroc") is not None for v in bl.values()): return "auroc", True return "mae", False def boards(): return cached("boards", 300, _load_boards) def _load_references(): """현업 도구를 이 부문 훈련 자료로 재학습해 얻은 성적. 다른 자료로 학습한 모델을 그대로 옮겨 재면, 훈련 자료의 차이가 방법의 차이로 읽힌다. 그래서 **방법만 가져오고 데이터는 이 부문 것을 쓴다.** 그래야 표에 오른 숫자가 "이 방법이 이 문제에서 어디까지 가는가"를 뜻한다. """ # 참조 방법은 여럿일 수 있다. 파일 하나에 방법 하나를 담고, 여기서 다 모은다. out = {} for p in sorted(glob.glob(os.path.join(DATA, "reference_*.json"))): try: d = json.load(io.open(p, encoding="utf-8")) except Exception: continue for e in d.get("entries", []): out.setdefault(e["board"], []).append( dict(e, method=d.get("method", "reference"), method_en=d.get("method_en") or d.get("method", "reference"), ref_note=d.get("note"), ref_note_en=d.get("note_en"))) return out def references(): return cached("refs", 300, _load_references) def categories(): def build(): cats = json.load(io.open(os.path.join(HERE, "categories.json"), encoding="utf-8")) bd = boards() for c in cats["categories"]: c["open"] = len([b for b in c["boards"] if b in bd]) c["board_cards"] = [bd[b] for b in c["boards"] if b in bd] return cats return cached("cats", 300, build) # ------------------------------------------------------------------ 원장 def _hdr(): return {"Authorization": "Bearer " + HF_TOKEN, "User-Agent": "LEADBOARD/1.0"} def submission_count(): """원장에 들어온 제출 수. **주최측 기준 제출도 포함한 전체다.** 화면 상단에 올리는 값이라 매 요청마다 원장을 훑으면 안 된다 - 트리 조회는 페이지당 1000건이고 부문이 46개다. 5분 캐시로 충분하다: 이 숫자는 사람이 제출할 때만 움직이고, 5분 늦게 보인다고 잘못된 값이 되지는 않는다. """ def build(): n, cursor = 0, "" for _ in range(20): # 20페이지 = 2만 건. 그 위는 지금 없다. url = (LEDGER_API + "/tree/main?recursive=1&limit=1000" + ("&cursor=" + cursor if cursor else "")) try: with urllib.request.urlopen( urllib.request.Request(url, headers=_hdr()), timeout=60) as r: rows = json.loads(r.read()) link = r.headers.get("Link") or "" except Exception: return None # 🔴 실패는 0 이 아니다. 모르는 것이다. n += sum(1 for x in rows if str(x.get("path", "")).startswith("submissions/") and str(x.get("path", "")).endswith(".json")) m = re.search(r"cursor=([^&>]+)", link) if 'rel="next"' in link else None if not m: break cursor = m.group(1) return n return cached("subcount", 300, build) def ledger_read(path, default=None): try: with urllib.request.urlopen(urllib.request.Request( "%s/%s" % (LEDGER_RAW, path), headers=_hdr()), timeout=60) as r: return json.loads(r.read()) except Exception: return default def ledger_write(path, obj, summary): blob = base64.b64encode(json.dumps(obj, ensure_ascii=False).encode()).decode() lines = [json.dumps({"key": "header", "value": {"summary": summary}}), json.dumps({"key": "file", "value": {"path": path, "content": blob, "encoding": "base64"}})] req = urllib.request.Request(LEDGER_API + "/commit/main", data=("\n".join(lines) + "\n").encode(), headers=dict(_hdr(), **{"Content-Type": "application/x-ndjson"})) with urllib.request.urlopen(req, timeout=180) as r: return json.loads(r.read()) # ------------------------------------------------------------------ 세션 def sign(v): return hmac.new(SESSION_KEY.encode(), v.encode(), hashlib.sha256).hexdigest()[:32] def set_session(resp, user): raw = json.dumps(user, ensure_ascii=False) b = base64.urlsafe_b64encode(raw.encode()).decode() resp.set_cookie(COOKIE, "%s.%s" % (b, sign(b)), max_age=86400 * 7, httponly=True, **COOKIE_KW) def who(req: Request): c = req.cookies.get(COOKIE) or "" if "." not in c: return None b, sg = c.rsplit(".", 1) if not hmac.compare_digest(sg, sign(b)): return None try: return json.loads(base64.urlsafe_b64decode(b.encode()).decode()) except Exception: return None # ------------------------------------------------------------------ 라우트 @app.get("/") def index(): return FileResponse(os.path.join(HERE, "index.html")) @app.get("/i18n.js") def i18n(): """문자열 사전. 화면 코드와 분리해 두어야 영어판이 조용히 뒤처지지 않는다.""" return FileResponse(os.path.join(HERE, "i18n.js"), media_type="application/javascript") @app.get("/api/categories") def api_categories(): c = categories() bd = boards() return {"spec": SPEC, "categories": c["categories"], "totals": {"planned": sum(x["planned"] for x in c["categories"]), "open": len(bd), "ledger": bool(HF_TOKEN)}} @app.get("/api/leaders") def api_leaders(): """부문마다 현재 1위 한 줄. 첫 화면에서 전체를 한눈에 보기 위한 것이다. 부문별로 순위표를 따로 부르면 왕복이 부문 수만큼 늘어난다. 여기서 한 번에 모은다. **참가 제출이 없으면 비워서 보낸다** - 기준선을 1위 자리에 앉히지 않는다. 기준선은 넘어야 할 선이지 우승자가 아니다. """ def build(): out = [] cover = {} # 누가 어느 부문에 제출했나 prize_lead = {} # 부문별 **참가자 중** 선두. 상금은 이것으로 센다. for name, b in boards().items(): rolled = ledger_read("leaderboard/%s.json" % name.lower(), {}) or {} ent = (rolled.get("entries") or []) # 상금 집계는 **주최측 제출을 아예 없는 것으로 보고** 계산한다. # 주최측을 수상 대상에서만 빼는 것으로는 부족했다 - 기준 제출이 46부문 # 1위를 전부 점유하자 외부 참가자의 1위가 0이 되어, 상금을 받을 수 있는 # 사람이 구조적으로 없어졌다. 표시상 1위는 그대로 두고(넘어야 할 눈금이 # 보여야 한다), 상금에서는 참가자 중 최상위를 그 부문의 선두로 센다. for e in ent: u = e.get("user") if u and not is_host(u): cover.setdefault(u, set()).add(name) lead_p = next((e for e in ent if e.get("user") and not is_host(e.get("user"))), None) if lead_p: prize_lead[name] = lead_p.get("user") top = ent[0] if ent else None nf = (b.get("noise_floor") or {}).get("sd_single") base = b.get("baselines") or {} mkey, mhb = primary_metric(b) vals = [v.get(mkey) for v in base.values() if v.get(mkey) is not None] row = {"board": name, "n_test": b.get("n_test"), "metric": mkey, "higher_better": mhb, # 주지표 위에서의 최선 기준선. 넘어야 할 선은 하나뿐이다. "baseline_best_score": (max(vals) if mhb else min(vals)) if vals else None, "blurb": b.get("blurb"), "blurb_en": b.get("blurb_en"), "split_grade": b.get("split_grade"), "answer_grade": b.get("answer_grade"), "noise_floor": nf, "entries": len(ent), "baseline_best": (min((v["mae"] for v in base.values() if v.get("mae") is not None), default=None)), "baseline_best_auroc": (max((v["auroc"] for v in base.values() if v.get("auroc") is not None), default=None)), "leader": None} if top: row["leader"] = {"method": top.get("method"), "user": top.get("user"), "host": is_host(top.get("user")), # score 는 그 부문 주지표 위의 값이다. mae 는 옛 표시부 호환. "score": top.get("score", top.get("mae")), "mae": top.get("mae"), "auroc": top.get("auroc"), "verified": bool(top.get("verified")), "leak": top.get("leak")} out.append(row) out.sort(key=lambda r: (r["leader"] is None, -(r["n_test"] or 0))) return {"rows": out, "cover": {u: sorted(s) for u, s in cover.items()}, "prize_lead": prize_lead} got = cached("leaders", 60, build) rows, cover = got["rows"], got["cover"] n_boards = len(rows) # 수상 후보. 주최측은 세지 않는다. # **자격 = 전 부문 제출.** 못 채운 사람도 표에는 올린다 - 몇 개 남았는지 보여야 # 채우러 온다. 감추면 왜 자기가 후보가 아닌지 알 수 없다. tally = {} for u in (got.get("prize_lead") or {}).values(): tally[u] = tally.get(u, 0) + 1 # 후보 목록. 1위를 가진 사람이 먼저 오고, 그 다음 참가 부문 수 순이다. # 1위가 없어도 목록에 남긴다 - 참가한 사람이 보여야 판이 살아 있는 것으로 읽힌다. standing = [] for u, bs in cover.items(): standing.append({"user": u, "firsts": tally.get(u, 0), "submitted": len(bs), "missing": 0, "eligible": True}) standing.sort(key=lambda x: (-x["firsts"], -x["submitted"], x["user"])) return {"spec": SPEC, "n": len(rows), "rows": rows, "submissions": submission_count(), "held": sum(1 for r in rows if r["leader"]), "prize": dict(PRIZE, n_boards=n_boards), "standing": standing, "host_accounts": sorted(HOST_ACCOUNTS)} @app.get("/api/board/{name}") def api_board(name: str): b = boards().get(name) if not b: raise HTTPException(404, "그런 부문이 없다") return b @app.get("/api/board/{name}/testset") def api_testset(name: str): """테스트셋. 구조만 나간다 - 라벨은 이 컨테이너에 존재하지 않는다.""" p = os.path.join(DATA, "%s_test.json" % name.lower()) if not os.path.exists(p): raise HTTPException(404, "테스트셋이 아직 없다") return FileResponse(p, media_type="application/json", filename="%s_test.json" % name.lower()) @app.get("/api/board/{name}/leaderboard") def api_leaderboard(name: str): """순위표. 기준선은 항상 포함된다 (3조). 참가자 항목은 워커가 굴려 놓은 것을 그대로 보여준다. 여기서 계산하지 않는다 - 계산하려면 정답이 있어야 하고, 정답은 여기 없다. """ b = boards().get(name) if not b: raise HTTPException(404, "그런 부문이 없다") rolled = cached("lb:" + name, 60, lambda: ledger_read("leaderboard/%s.json" % name.lower(), {"entries": [], "updated": None})) or {} rows = [] for k, v in (b.get("baselines") or {}).items(): rows.append({"method": k, "user": "—", "kind": "baseline", "mae": v["mae"], "auroc": v["auroc"], "prauc": v["prauc"]}) # 현업 도구 참조 항목. 기준선과 참가 제출 사이에 놓는다 - 학습하지 않은 선도 아니고 # 이번 회차의 참가자도 아니다. 참가자가 자기 위치를 가늠할 세 번째 좌표다. for r in references().get(name, []): rows.append({"method": r["method"], "method_en": r.get("method_en"), "user": "—", "kind": "reference", "mae": r.get("mae"), "auroc": r.get("auroc"), "prauc": r.get("prauc"), "leak": r.get("leak"), "note": r.get("ref_note"), "note_en": r.get("ref_note_en")}) for e in rolled.get("entries", []): rows.append(dict(e, kind=e.get("kind", "entry"))) # 부문 주지표로 세운다. 분류 전용 부문에서 mae 로 세우면 전부 동률이 된다. if b.get("metric") == "auroc": rows.sort(key=lambda r: -(r.get("auroc") or 0)) else: rows.sort(key=lambda r: (r.get("mae") is None, r.get("mae") or 9e9)) nf = (b.get("noise_floor") or {}).get("sd_single") best = rows[0].get("mae") if rows else None # 4조: 최고점에서 잡음 바닥 안에 든 항목은 같은 계단으로 묶는다. for r in rows: r["within_noise"] = bool(nf and best is not None and r.get("mae") is not None and r["mae"] - best < nf) rank = 0 for r in rows: # 순위는 참가 제출에만 매긴다. 기준선과 참조 도구는 표에 서되 등수를 갖지 않는다 - # 우리가 올린 것이 1위 자리를 차지하면 참가자에게 겨룰 자리가 없다. if r["kind"] in ("baseline", "reference"): r["rank"] = None else: rank += 1 r["rank"] = rank return {"board": name, "noise_floor": nf, "rows": rows, "updated": rolled.get("updated"), "note": "잡음 바닥 안에 든 항목은 순위 차이로 주장하지 않는다 (4조)"} # ------------------------------------------------------------------ 로그인 @app.get("/login") def login(request: Request): if not (OAUTH_ID and OAUTH_SECRET): return _err_page("이 시험대에 로그인이 아직 구성되지 않았습니다.") nxt = request.query_params.get("next", "/") st = base64.urlsafe_b64encode(json.dumps({"n": nxt, "r": secrets.token_hex(8)}).encode()).decode() q = urllib.parse.urlencode({ "client_id": OAUTH_ID, "redirect_uri": _redirect(request), "response_type": "code", "scope": "openid profile", "state": st}) return RedirectResponse("%s/oauth/authorize?%s" % (OAUTH_ISS, q)) def _redirect(request: Request): """플랫폼이 등록해 주는 콜백 주소는 **/auth/callback** 이다. 여기를 /auth 로 두면 토큰 교환에서 redirect_uri 불일치로 거부되고, 그 예외가 그대로 500 이 되어 화면 전체가 죽는다. 로그인 한 번 눌렀다가 사이트가 사라진다. """ if SPACE_HOST: return "https://%s/auth/callback" % SPACE_HOST return str(request.base_url).rstrip("/") + "/auth/callback" def _err_page(msg, detail=""): """로그인이 실패해도 화면은 살아 있어야 한다. 흰 배경에 Internal Server Error 만 남으면 이용자는 사이트가 죽은 줄 안다.""" return Response( "
" "%s
%s" "" % (msg, ("%s