"""chartbench 휴먼 이밸: Qwen vs Opus blind A/B (4축). - 접속: 공유 링크의 ?key= 가 ACCESS_KEY 와 일치해야 평가 가능 (Space는 public이지만 키 없으면 진입 불가) - 할당: 미완료 아이템 중 최근 10분 내 타인에게 할당되지 않은 것을 랜덤 지급. 1명 제출 = 그 아이템 완료. - 저장: 제출 1건 = results repo 에 JSON 1파일 (rater / 할당·제출 시각 / 소요초 포함). 재시작 시 repo 로 완료 목록 복원. - 같은 이름 = 같은 평가자로 집계 (분석 시 이름 기준 dedup). """ import json, os, time, uuid, random, threading from datetime import datetime, timezone, timedelta from pathlib import Path import gradio as gr from huggingface_hub import HfApi ACCESS_KEY = os.environ.get("ACCESS_KEY", "devkey") RESULTS_REPO = os.environ.get("RESULTS_REPO", "") HF_TOKEN = os.environ.get("HF_TOKEN", "") RESERVE_TTL = 600 KST = timezone(timedelta(hours=9)) ROOT = Path(__file__).parent ITEMS = [json.loads(l) for l in (ROOT/"items.jsonl").open()] BY_ID = {it["item_id"]: it for it in ITEMS} HINT = '
{}
' AXES = [ ("query_following", "**1. 쿼리 요구사항 반영** — 어느 차트가 질문이 요구한 바를 더 잘 반영했나요?"), ("chart_quality", "**2. 가독성·차트 품질** — 어느 차트가 더 읽기 좋고 적절한 형태인가요?"), ("data_fidelity", "**3. Data 반영도** — 어느 차트가 기반 데이터를 더 정확히 반영했나요?" + HINT.format("두 차트의 수치가 유사해 보이면 '비슷함'을 눌러주세요. 크게 다르면 하단 토글의 원본 표를 확인해 판단하시고, 표 자체가 이상하거나 둘 다 잘못 반영했으면 '비슷함'을 눌러주세요.")), ("overall", "**4. 전반적 선호** — 종합적으로 어느 차트를 선호하시나요?"), ] CHOICES = [("차트 A", "A"), ("차트 B", "B"), ("비슷함 / 판단 어려움", "Tie")] GUIDE = """ ### 안내 1. **이름을 입력하고 시작**을 누르면 선호 판단 문제가 무작위로 제시됩니다. *(익명으로 작성하셔도 됩니다)* 2. 각 문제는 **질문(쿼리) 1개 + 차트 2개(A/B)** 로 구성되며, 차트의 기반 데이터(원본 표)는 하단에 접혀 있습니다. 3. 판단 항목은 **4가지** — 쿼리 요구사항 반영 · 가독성 · Data 반영도 · 전반적 선호 — 각각 **A / B / 비슷함** 중 선택합니다. 4. 제출하면 그 문제는 완료되고 다음 문제가 나옵니다. **인당 10개 이상 해주시면 감사하겠습니다.** *(바쁘시다면 더 적게 해주셔도 큰 도움이 됩니다!)* 중간에 그냥 창을 닫아도 됩니다. 5. 가볍게 보고 **느껴지는 대로** 선택하시면 됩니다. Data 반영도는 표를 일일이 대조하기보다, 두 차트의 수치가 크게 다를 때만 표를 참고하는 정도면 충분합니다. """ CSS = """ .guide-box {font-size: 1.15rem; line-height: 1.75;} .guide-box li {margin-bottom: 6px;} .axis-hint {color: var(--body-text-color-subdued); font-size: 0.92em; margin-top: 2px;} .query-card {font-size: 1.25rem; line-height: 1.55; background: var(--block-background-fill); border: 1px solid var(--border-color-primary); border-left: 6px solid #4f7cff; border-radius: 10px; padding: 14px 18px; margin: 4px 0 10px;} .axis-block .wrap {gap: 4px;} footer {display: none !important;} """ api = HfApi(token=HF_TOKEN) if HF_TOKEN else None _lock = threading.Lock() _completed: set[str] = set() _reserved: dict[str, float] = {} _rater_counts: dict[str, int] = {} def _load_completed(): """결과 repo 스냅샷에서 완료 아이템과 평가자별 누적 수를 복원한다.""" if not (api and RESULTS_REPO): return try: from huggingface_hub import snapshot_download local = snapshot_download(RESULTS_REPO, repo_type="dataset", token=HF_TOKEN, allow_patterns=["results/*.json"]) for p in Path(local).glob("results/*.json"): try: rec = json.loads(p.read_text()) except Exception: continue if rec.get("item_id"): _completed.add(rec["item_id"]) r = (rec.get("rater") or "").strip() if r: _rater_counts[r] = _rater_counts.get(r, 0) + 1 except Exception as e: print("results repo load failed:", e) _load_completed() print(f"items={len(ITEMS)} completed={len(_completed)}") def _now_iso() -> str: return datetime.now(KST).isoformat(timespec="seconds") def _pick_item() -> str | None: now = time.time() with _lock: for k, t in list(_reserved.items()): if now - t > RESERVE_TTL: del _reserved[k] pool = [it["item_id"] for it in ITEMS if it["item_id"] not in _completed and it["item_id"] not in _reserved] if not pool: pool = [it["item_id"] for it in ITEMS if it["item_id"] not in _completed] if not pool: return None item_id = random.choice(pool) _reserved[item_id] = now return item_id def _save(rater: str, item_id: str, answers: dict, assigned_at: str, assigned_ts: float): rec = {"rater": rater, "item_id": item_id, "answers": answers, "assigned_at": assigned_at, "submitted_at": _now_iso(), "elapsed_s": round(time.time() - assigned_ts, 1) if assigned_ts else None, "id": uuid.uuid4().hex[:8]} # 저장이 성공한 뒤에만 완료 처리한다: 업로드 실패 시 예외를 올려 사용자에게 # 재제출을 안내하고, 아이템은 미완료로 남는다 (응답 유실 방지). if api and RESULTS_REPO: num = item_id.split("_")[1] api.upload_file(path_or_fileobj=json.dumps(rec, ensure_ascii=False).encode(), path_in_repo=f"results/r_{num}_{rec['id']}.json", repo_id=RESULTS_REPO, repo_type="dataset", commit_message=f"rating {item_id} by {rater}") else: with (ROOT/"local_results.jsonl").open("a") as f: f.write(json.dumps(rec, ensure_ascii=False)+"\n") with _lock: _completed.add(item_id) _reserved.pop(item_id, None) _rater_counts[rater] = _rater_counts.get(rater, 0) + 1 def _progress(my_count: int) -> str: return f"전체 진행 **{len(_completed)} / {len(ITEMS)}**  ·  내가 평가한 수 **{my_count}**" def _render_item(item_id: str): it = BY_ID[item_id] return (f'
질문
{it["query"]}
', str(ROOT/"images"/f"{item_id}_A.png"), str(ROOT/"images"/f"{item_id}_B.png"), it["table_markdown"]) N_START_OUT = 12 # gate, panel, query, imgA, imgB, table, progress, rater, item, assigned_at, assigned_ts, my_count def start(rater, request: gr.Request): if (request.query_params.get("key") or "") != ACCESS_KEY: raise gr.Error("접근 키가 없습니다. 공유받은 링크로 접속하세요.") rater = (rater or "").strip() if not rater: gr.Warning("이름(또는 익명 코드)을 입력해주세요.") return tuple(gr.skip() for _ in range(N_START_OUT)) my_count = _rater_counts.get(rater, 0) item_id = _pick_item() if item_id is None: gr.Info("모든 문제 평가가 끝났습니다. 감사합니다!") return (gr.update(visible=True), gr.update(visible=False), "", None, None, "", "🎉 모든 문제가 완료되었습니다!", rater, "", "", 0.0, my_count) q, a, b, t = _render_item(item_id) return (gr.update(visible=False), gr.update(visible=True), q, a, b, t, _progress(my_count), rater, item_id, _now_iso(), time.time(), my_count) N_SUB_OUT = 11 # query, imgA, imgB, table, progress, item, assigned_at, assigned_ts, my_count, + 4 radios -> actually computed below def submit(rater, item_id, assigned_at, assigned_ts, my_count, c1, c2, c3, c4, request: gr.Request): choices = (c1, c2, c3, c4) n_out = 9 + len(AXES) if (request.query_params.get("key") or "") != ACCESS_KEY: raise gr.Error("접근 키가 없습니다.") if not item_id: gr.Warning("할당된 문제가 없습니다. 새로고침 후 다시 시작해주세요.") return tuple(gr.skip() for _ in range(n_out)) missing = [f"{i+1}번" for i, c in enumerate(choices) if c is None] if missing: gr.Warning("모든 항목을 선택해주세요. 미선택: " + ", ".join(missing)) return tuple(gr.skip() for _ in range(n_out)) try: _save(rater, item_id, {k: c for (k, _), c in zip(AXES, choices)}, assigned_at, assigned_ts) except Exception as e: print("save failed:", type(e).__name__, e) gr.Warning("저장에 실패했습니다 (네트워크 문제일 수 있습니다). 잠시 후 '제출하고 다음'을 다시 눌러주세요 — 선택은 유지됩니다.") with _lock: _reserved[item_id] = time.time() # 이 사용자 몫으로 예약 연장 return tuple(gr.skip() for _ in range(n_out)) my_count = (my_count or 0) + 1 nxt = _pick_item() if nxt is None: return ("🎉 모든 문제가 완료되었습니다. 참여해주셔서 감사합니다!", None, None, "", _progress(my_count), "", "", 0.0, my_count, *[None]*len(AXES)) q, a, b, t = _render_item(nxt) return (q, a, b, t, _progress(my_count), nxt, _now_iso(), time.time(), my_count, *[None]*len(AXES)) with gr.Blocks(title="차트 선호 평가") as demo: gr.Markdown("# 📊 차트 생성 선호 평가 (Blind A/B)\n같은 질문·같은 데이터로 두 시스템이 만든 차트입니다. 항목별로 **더 나은 쪽**을 골라주세요. A/B 배치는 문제마다 무작위입니다.") with gr.Group(visible=True) as gate: rater_in = gr.Textbox(label="이름 (또는 익명 코드)", placeholder="예: 박OO") start_btn = gr.Button("평가 시작", variant="primary") gr.Markdown(GUIDE, elem_classes="guide-box") with gr.Group(visible=False) as panel: prog = gr.Markdown() qhtml = gr.HTML() with gr.Row(): img_a = gr.Image(label="차트 A", type="filepath", height=440) img_b = gr.Image(label="차트 B", type="filepath", height=440) radios = [] for _, axis_md in AXES: gr.Markdown(axis_md) radios.append(gr.Radio(CHOICES, show_label=False, elem_classes="axis-block")) sub_btn = gr.Button("제출하고 다음 →", variant="primary", size="lg") with gr.Accordion("원본 표 펼쳐보기 (Data 반영도 참고용)", open=False): tbl = gr.Markdown() rater_st = gr.State(""); item_st = gr.State(""); at_st = gr.State(""); ts_st = gr.State(0.0); cnt_st = gr.State(0) start_btn.click(start, [rater_in], [gate, panel, qhtml, img_a, img_b, tbl, prog, rater_st, item_st, at_st, ts_st, cnt_st]) sub_btn.click(submit, [rater_st, item_st, at_st, ts_st, cnt_st, *radios], [qhtml, img_a, img_b, tbl, prog, item_st, at_st, ts_st, cnt_st, *radios]) if __name__ == "__main__": demo.launch(css=CSS)