"""chartbench 휴먼 이밸: Qwen vs Opus blind A/B (4축).
- 접속: 공유 링크의 ?key= 가 ACCESS_KEY 와 일치해야 평가 가능 (Space는 public이지만 키 없으면 진입 불가)
- 할당: 미완료 아이템 중 최근 10분 내 타인에게 할당되지 않은 것을 랜덤 지급. 1명 제출 = 그 아이템 완료.
- 저장: 제출 1건 = results repo 에 JSON 1파일 (rater / 할당·제출 시각 / 소요초 포함). 재시작 시 repo 로 완료 목록 복원.
- 같은 이름 = 같은 평가자로 집계 (분석 시 이름 기준 dedup).
"""
import json, os, time, uuid, random, threading
from datetime import datetime, timezone, timedelta
from pathlib import Path
import gradio as gr
from huggingface_hub import HfApi
ACCESS_KEY = os.environ.get("ACCESS_KEY", "devkey")
RESULTS_REPO = os.environ.get("RESULTS_REPO", "")
HF_TOKEN = os.environ.get("HF_TOKEN", "")
RESERVE_TTL = 600
KST = timezone(timedelta(hours=9))
ROOT = Path(__file__).parent
ITEMS = [json.loads(l) for l in (ROOT/"items.jsonl").open()]
BY_ID = {it["item_id"]: it for it in ITEMS}
HINT = '
{}
'
AXES = [
("query_following",
"**1. 쿼리 요구사항 반영** — 어느 차트가 질문이 요구한 바를 더 잘 반영했나요?"),
("chart_quality",
"**2. 가독성·차트 품질** — 어느 차트가 더 읽기 좋고 적절한 형태인가요?"),
("data_fidelity",
"**3. Data 반영도** — 어느 차트가 기반 데이터를 더 정확히 반영했나요?"
+ HINT.format("두 차트의 수치가 유사해 보이면 '비슷함'을 눌러주세요. 크게 다르면 하단 토글의 원본 표를 확인해 판단하시고, 표 자체가 이상하거나 둘 다 잘못 반영했으면 '비슷함'을 눌러주세요.")),
("overall",
"**4. 전반적 선호** — 종합적으로 어느 차트를 선호하시나요?"),
]
CHOICES = [("차트 A", "A"), ("차트 B", "B"), ("비슷함 / 판단 어려움", "Tie")]
GUIDE = """
### 안내
1. **이름을 입력하고 시작**을 누르면 선호 판단 문제가 무작위로 제시됩니다. *(익명으로 작성하셔도 됩니다)*
2. 각 문제는 **질문(쿼리) 1개 + 차트 2개(A/B)** 로 구성되며, 차트의 기반 데이터(원본 표)는 하단에 접혀 있습니다.
3. 판단 항목은 **4가지** — 쿼리 요구사항 반영 · 가독성 · Data 반영도 · 전반적 선호 — 각각 **A / B / 비슷함** 중 선택합니다.
4. 제출하면 그 문제는 완료되고 다음 문제가 나옵니다. **인당 10개 이상 해주시면 감사하겠습니다.** *(바쁘시다면 더 적게 해주셔도 큰 도움이 됩니다!)* 중간에 그냥 창을 닫아도 됩니다.
5. 가볍게 보고 **느껴지는 대로** 선택하시면 됩니다. Data 반영도는 표를 일일이 대조하기보다, 두 차트의 수치가 크게 다를 때만 표를 참고하는 정도면 충분합니다.
"""
CSS = """
.guide-box {font-size: 1.15rem; line-height: 1.75;}
.guide-box li {margin-bottom: 6px;}
.axis-hint {color: var(--body-text-color-subdued); font-size: 0.92em; margin-top: 2px;}
.query-card {font-size: 1.25rem; line-height: 1.55; background: var(--block-background-fill);
border: 1px solid var(--border-color-primary); border-left: 6px solid #4f7cff;
border-radius: 10px; padding: 14px 18px; margin: 4px 0 10px;}
.axis-block .wrap {gap: 4px;}
footer {display: none !important;}
"""
api = HfApi(token=HF_TOKEN) if HF_TOKEN else None
_lock = threading.Lock()
_completed: set[str] = set()
_reserved: dict[str, float] = {}
_rater_counts: dict[str, int] = {}
def _load_completed():
"""결과 repo 스냅샷에서 완료 아이템과 평가자별 누적 수를 복원한다."""
if not (api and RESULTS_REPO):
return
try:
from huggingface_hub import snapshot_download
local = snapshot_download(RESULTS_REPO, repo_type="dataset", token=HF_TOKEN,
allow_patterns=["results/*.json"])
for p in Path(local).glob("results/*.json"):
try:
rec = json.loads(p.read_text())
except Exception:
continue
if rec.get("item_id"):
_completed.add(rec["item_id"])
r = (rec.get("rater") or "").strip()
if r:
_rater_counts[r] = _rater_counts.get(r, 0) + 1
except Exception as e:
print("results repo load failed:", e)
_load_completed()
print(f"items={len(ITEMS)} completed={len(_completed)}")
def _now_iso() -> str:
return datetime.now(KST).isoformat(timespec="seconds")
def _pick_item() -> str | None:
now = time.time()
with _lock:
for k, t in list(_reserved.items()):
if now - t > RESERVE_TTL:
del _reserved[k]
pool = [it["item_id"] for it in ITEMS
if it["item_id"] not in _completed and it["item_id"] not in _reserved]
if not pool:
pool = [it["item_id"] for it in ITEMS if it["item_id"] not in _completed]
if not pool:
return None
item_id = random.choice(pool)
_reserved[item_id] = now
return item_id
def _save(rater: str, item_id: str, answers: dict, assigned_at: str, assigned_ts: float):
rec = {"rater": rater, "item_id": item_id, "answers": answers,
"assigned_at": assigned_at, "submitted_at": _now_iso(),
"elapsed_s": round(time.time() - assigned_ts, 1) if assigned_ts else None,
"id": uuid.uuid4().hex[:8]}
# 저장이 성공한 뒤에만 완료 처리한다: 업로드 실패 시 예외를 올려 사용자에게
# 재제출을 안내하고, 아이템은 미완료로 남는다 (응답 유실 방지).
if api and RESULTS_REPO:
num = item_id.split("_")[1]
api.upload_file(path_or_fileobj=json.dumps(rec, ensure_ascii=False).encode(),
path_in_repo=f"results/r_{num}_{rec['id']}.json",
repo_id=RESULTS_REPO, repo_type="dataset",
commit_message=f"rating {item_id} by {rater}")
else:
with (ROOT/"local_results.jsonl").open("a") as f:
f.write(json.dumps(rec, ensure_ascii=False)+"\n")
with _lock:
_completed.add(item_id)
_reserved.pop(item_id, None)
_rater_counts[rater] = _rater_counts.get(rater, 0) + 1
def _progress(my_count: int) -> str:
return f"전체 진행 **{len(_completed)} / {len(ITEMS)}** · 내가 평가한 수 **{my_count}**"
def _render_item(item_id: str):
it = BY_ID[item_id]
return (f'질문
{it["query"]}
',
str(ROOT/"images"/f"{item_id}_A.png"), str(ROOT/"images"/f"{item_id}_B.png"),
it["table_markdown"])
N_START_OUT = 12 # gate, panel, query, imgA, imgB, table, progress, rater, item, assigned_at, assigned_ts, my_count
def start(rater, request: gr.Request):
if (request.query_params.get("key") or "") != ACCESS_KEY:
raise gr.Error("접근 키가 없습니다. 공유받은 링크로 접속하세요.")
rater = (rater or "").strip()
if not rater:
gr.Warning("이름(또는 익명 코드)을 입력해주세요.")
return tuple(gr.skip() for _ in range(N_START_OUT))
my_count = _rater_counts.get(rater, 0)
item_id = _pick_item()
if item_id is None:
gr.Info("모든 문제 평가가 끝났습니다. 감사합니다!")
return (gr.update(visible=True), gr.update(visible=False), "", None, None, "",
"🎉 모든 문제가 완료되었습니다!", rater, "", "", 0.0, my_count)
q, a, b, t = _render_item(item_id)
return (gr.update(visible=False), gr.update(visible=True), q, a, b, t,
_progress(my_count), rater, item_id, _now_iso(), time.time(), my_count)
N_SUB_OUT = 11 # query, imgA, imgB, table, progress, item, assigned_at, assigned_ts, my_count, + 4 radios -> actually computed below
def submit(rater, item_id, assigned_at, assigned_ts, my_count, c1, c2, c3, c4, request: gr.Request):
choices = (c1, c2, c3, c4)
n_out = 9 + len(AXES)
if (request.query_params.get("key") or "") != ACCESS_KEY:
raise gr.Error("접근 키가 없습니다.")
if not item_id:
gr.Warning("할당된 문제가 없습니다. 새로고침 후 다시 시작해주세요.")
return tuple(gr.skip() for _ in range(n_out))
missing = [f"{i+1}번" for i, c in enumerate(choices) if c is None]
if missing:
gr.Warning("모든 항목을 선택해주세요. 미선택: " + ", ".join(missing))
return tuple(gr.skip() for _ in range(n_out))
try:
_save(rater, item_id, {k: c for (k, _), c in zip(AXES, choices)}, assigned_at, assigned_ts)
except Exception as e:
print("save failed:", type(e).__name__, e)
gr.Warning("저장에 실패했습니다 (네트워크 문제일 수 있습니다). 잠시 후 '제출하고 다음'을 다시 눌러주세요 — 선택은 유지됩니다.")
with _lock:
_reserved[item_id] = time.time() # 이 사용자 몫으로 예약 연장
return tuple(gr.skip() for _ in range(n_out))
my_count = (my_count or 0) + 1
nxt = _pick_item()
if nxt is None:
return ("🎉 모든 문제가 완료되었습니다. 참여해주셔서 감사합니다!", None, None, "",
_progress(my_count), "", "", 0.0, my_count, *[None]*len(AXES))
q, a, b, t = _render_item(nxt)
return (q, a, b, t, _progress(my_count), nxt, _now_iso(), time.time(), my_count, *[None]*len(AXES))
with gr.Blocks(title="차트 선호 평가") as demo:
gr.Markdown("# 📊 차트 생성 선호 평가 (Blind A/B)\n같은 질문·같은 데이터로 두 시스템이 만든 차트입니다. 항목별로 **더 나은 쪽**을 골라주세요. A/B 배치는 문제마다 무작위입니다.")
with gr.Group(visible=True) as gate:
rater_in = gr.Textbox(label="이름 (또는 익명 코드)", placeholder="예: 박OO")
start_btn = gr.Button("평가 시작", variant="primary")
gr.Markdown(GUIDE, elem_classes="guide-box")
with gr.Group(visible=False) as panel:
prog = gr.Markdown()
qhtml = gr.HTML()
with gr.Row():
img_a = gr.Image(label="차트 A", type="filepath", height=440)
img_b = gr.Image(label="차트 B", type="filepath", height=440)
radios = []
for _, axis_md in AXES:
gr.Markdown(axis_md)
radios.append(gr.Radio(CHOICES, show_label=False, elem_classes="axis-block"))
sub_btn = gr.Button("제출하고 다음 →", variant="primary", size="lg")
with gr.Accordion("원본 표 펼쳐보기 (Data 반영도 참고용)", open=False):
tbl = gr.Markdown()
rater_st = gr.State(""); item_st = gr.State(""); at_st = gr.State(""); ts_st = gr.State(0.0); cnt_st = gr.State(0)
start_btn.click(start, [rater_in],
[gate, panel, qhtml, img_a, img_b, tbl, prog, rater_st, item_st, at_st, ts_st, cnt_st])
sub_btn.click(submit, [rater_st, item_st, at_st, ts_st, cnt_st, *radios],
[qhtml, img_a, img_b, tbl, prog, item_st, at_st, ts_st, cnt_st, *radios])
if __name__ == "__main__":
demo.launch(css=CSS)