"""판단 전용 모델(Strands Decider 2B, Cloudflare Clef-Flash 9B)을 같은 문제로 잰다. 두 모델 모두 Jev/SystemOne 요청 모양({state, questions})을 받는다. 그래서 같은 요청을 그대로 보낸다. - 정답률: testset.py 의 100문제를 한국어·영어로(문제 하나에 질문 하나) - 걸린 시간: 문제 하나씩 보낼 때의 시간. 전체를 --repeat 번 돌려 평균·중앙값·p95(첫 번째 호출은 따로 적는다) - 메모리: 모델을 올린 뒤와 측정 중의 VRAM(PyTorch 가 잡은 양, nvidia-smi 의 GPU 전체), 시스템 램 - 여러 질문 묶기: 같은 글에 질문 3개를 한 번에 물을 때의 시간 사용(저장소 밖 실행 환경의 파이썬으로): python bench_decider.py strands [--repeat 5] [--device cuda] python bench_decider.py clef-flash python bench_decider.py d1 [--compile] # Liquid AI d1-3B(transformers 5.14 이상) 결과: results/<날짜>__<모델>.json """ import argparse, json, socket, statistics, subprocess, sys, time from pathlib import Path import psutil import torch sys.stdout.reconfigure(encoding="utf-8") HERE = Path(__file__).resolve().parent sys.path.insert(0, str(HERE)) import testset, hardset # noqa: E402 PC = {"woojungho_28590": "5090", "woojungho_1660": "5060ti"}.get(socket.gethostname().lower(), socket.gethostname()) MODELS = {"strands": "StrandsAgents/strands-decider-2B-hobson-v19", "clef-flash": "Cloudflare/clef-flash", "d1": "LiquidAI/d1-3B"} D1_REV = "051bcc464b01b9f92942b364d9586b0ef5912432" # 코드를 읽어 본 커밋(trust_remote_code) ap = argparse.ArgumentParser() ap.add_argument("model", choices=MODELS) ap.add_argument("--repeat", type=int, default=5) ap.add_argument("--device", default="cuda") ap.add_argument("--tag", default="") ap.add_argument("--fp32", action="store_true", help="d1 만: 가중치를 fp32 로 올린다") ap.add_argument("--compile", action="store_true", help="d1 만: model.compile(mode='reduce-overhead')") ap.add_argument("--set", dest="qset", choices=["basic", "hard"], default="basic", help="basic=testset.py 100문제, hard=hardset.py 44문제") args = ap.parse_args() TASKS, records = (testset.TASKS, testset.records) if args.qset == "basic" else (hardset.TASKS, hardset.records) def smi(): o = subprocess.run(["nvidia-smi", "--query-gpu=memory.used,utilization.gpu,power.draw,temperature.gpu", "--format=csv,noheader,nounits"], capture_output=True, text=True).stdout.strip().split(",") return [float(x) for x in o] def sync(): if args.device.startswith("cuda"): torch.cuda.synchronize() def question_of(r): if r["type"] == "noul": return {"type": "noul", "instructions": r["question"]} return {"type": "choice", "instructions": r["question"], "criteria": r["options"]} def answer_of(r, ans): """응답 → (고른 보기, 확신도)""" if r["type"] == "noul": p = ans["noul"] return ("yes" if p >= 0.5 else "no"), (p if p >= 0.5 else 1 - p) return ans["choice"], ans["confidence"] ram0 = psutil.virtual_memory().used idle = smi() t0 = time.perf_counter() if args.model == "strands": from strands_decider.infer import load_engine from strands_decider.schema import SystemOneRequest engine = load_engine(MODELS["strands"], device=args.device) def ask(state, questions): resp = engine.evaluate(SystemOneRequest(state=state, questions=questions)) return json.loads(resp.model_dump_json())["answers"] elif args.model == "d1": from transformers import AutoModel from huggingface_hub import snapshot_download model = AutoModel.from_pretrained(snapshot_download(MODELS["d1"], revision=D1_REV), trust_remote_code=True, dtype=torch.float32 if (args.device == "cpu" or args.fp32) else torch.bfloat16).to(args.device) # 윈도우용 PyTorch 에는 flash attention 이 빠져 있다("USE_FLASH_ATTENTION was not enabled for build"). # 모델 코드(hybrid.py)가 CPU·맥에서 쓰는 대체 계산(_explicit, fp32)을 GPU 에서도 타게 한다. --fp32 면 손대지 않아도 그 길로 간다 flash = "native" if args.device.startswith("cuda") and not args.fp32: try: torch.ops.aten._flash_attention_forward ok = "USE_FLASH_ATTENTION" not in torch.__config__.show() or "USE_FLASH_ATTENTION=ON" in torch.__config__.show() except Exception: ok = False if sys.platform == "win32" or not ok: for name, mod in list(sys.modules.items()): if name.endswith(".hybrid") and hasattr(mod, "_fused"): mod._fused = lambda q: False flash = "off(explicit fp32 attention)" elif args.fp32: flash = "off(fp32 weights)" if args.compile: model.compile(mode="reduce-overhead") def ask(state, questions): return model.system_one(state, questions)["answers"] else: from huggingface_hub import snapshot_download path = snapshot_download(MODELS["clef-flash"]) sys.path.insert(0, path) from joint_schema_model import load_release_model, systemone model, processor = load_release_model(path, device=args.device) def ask(state, questions): return systemone(model, processor, {"model": "clef-flash", "state": state, "questions": questions})["answers"] sync() load_s = time.perf_counter() - t0 after = smi() R = {"started": time.strftime("%Y-%m-%d %H:%M:%S"), "pc": PC, "model": args.model, "repo": MODELS[args.model], "device": args.device, "compile": args.compile, "fp32": args.fp32, "flash": globals().get("flash"), "versions": {"torch": torch.__version__, "transformers": __import__("transformers").__version__, "python": sys.version.split()[0]}, "gpu": torch.cuda.get_device_name(0) if args.device.startswith("cuda") else "cpu", "load_s": round(load_s, 2), "idle_vram_mib": idle[0], "vram_after_load_mib": after[0], "torch_alloc_after_load_mib": round(torch.cuda.memory_allocated() / 2**20) if args.device.startswith("cuda") else None, "ram_before_gb": round(ram0 / 2**30, 2), "ram_after_load_gb": round(psutil.virtual_memory().used / 2**30, 2), "qset": args.qset, "langs": {}} print(f"load {load_s:.1f}s vram {idle[0]:.0f} -> {after[0]:.0f} MiB", flush=True) for lang in ("ko", "en"): recs = records(lang) # 첫 호출(준비 시간이 섞인다)은 따로 잰다 t = time.perf_counter(); ask(recs[0]["state"], {"q": question_of(recs[0])}); sync() first_ms = (time.perf_counter() - t) * 1000 rows, lat = [], [] if args.device.startswith("cuda"): torch.cuda.reset_peak_memory_stats() samples = [] for rep in range(args.repeat): for i, r in enumerate(recs): t = time.perf_counter() ans = ask(r["state"], {"q": question_of(r)})["q"] sync() ms = (time.perf_counter() - t) * 1000 lat.append(ms) if rep == 0: pred, conf = answer_of(r, ans) rows.append({"id": r["id"], "task": r["task"], "gold": r["gold"], "pred": pred, "conf": round(conf, 4), "ok": pred == r["gold"], "probs": ans.get("probabilities") or {"yes": ans.get("noul")}}) if i % 25 == 0: samples.append(smi()) acc = {task: [sum(x["ok"] for x in rows if x["task"] == task), sum(1 for x in rows if x["task"] == task)] for task in TASKS} lat_s = sorted(lat) wrong_conf = [x["conf"] for x in rows if not x["ok"]] right_conf = [x["conf"] for x in rows if x["ok"]] R["langs"][lang] = { "correct": sum(x["ok"] for x in rows), "total": len(rows), "by_task": acc, "first_call_ms": round(first_ms, 1), "calls": len(lat), "ms_mean": round(statistics.mean(lat), 2), "ms_median": round(statistics.median(lat), 2), "ms_p95": round(lat_s[int(len(lat_s) * 0.95) - 1], 2), "ms_min": round(lat_s[0], 2), "ms_max": round(lat_s[-1], 2), "ms_by_repeat": [round(statistics.mean(lat[k * len(recs):(k + 1) * len(recs)]), 2) for k in range(args.repeat)], "conf_right_mean": round(statistics.mean(right_conf), 4) if right_conf else None, "conf_wrong_mean": round(statistics.mean(wrong_conf), 4) if wrong_conf else None, "vram_peak_mib": max(s[0] for s in samples), "gpu_util_avg": round(statistics.mean(s[1] for s in samples), 1), "power_avg_w": round(statistics.mean(s[2] for s in samples), 1), "torch_peak_mib": round(torch.cuda.max_memory_allocated() / 2**20) if args.device.startswith("cuda") else None, "ram_gb": round(psutil.virtual_memory().used / 2**30, 2), "rows": rows, } d = R["langs"][lang] print(lang, f'{d["correct"]}/{d["total"]}', d["by_task"], f'mean {d["ms_mean"]}ms median {d["ms_median"]} p95 {d["ms_p95"]} first {d["first_call_ms"]}', flush=True) # 같은 글에 질문 3개를 한 번에(문의 글 25개: 팀·긴급·감정). 기본 문제에서만 잰다 multi = [] for lang in (("ko", "en") if args.qset == "basic" else ()): recs = [r for r in records(lang) if r["task"] == "inquiry"] urgent, senti = TASKS["urgent"], TASKS["sentiment"] qs = lambda r: {"team": question_of(r), "urgent": {"type": "noul", "instructions": urgent["q"][lang]}, "sentiment": {"type": "choice", "instructions": senti["q"][lang], "criteria": {k: v[lang] for k, v in senti["options"].items()}}} ask(recs[0]["state"], qs(recs[0])); sync() lat3, lat1 = [], [] for _ in range(3): for r in recs: t = time.perf_counter(); ask(r["state"], qs(r)); sync(); lat3.append((time.perf_counter() - t) * 1000) t = time.perf_counter(); ask(r["state"], {"team": question_of(r)}); sync(); lat1.append((time.perf_counter() - t) * 1000) multi.append({"lang": lang, "three_questions_ms": round(statistics.mean(lat3), 2), "one_question_ms": round(statistics.mean(lat1), 2)}) R["multi"] = multi print("multi", multi, flush=True) R["finished"] = time.strftime("%Y-%m-%d %H:%M:%S") out = HERE / "results"; out.mkdir(exist_ok=True) name = f"{time.strftime('%Y-%m-%d')}_{PC}_{args.model}{'' if args.qset == 'basic' else '_hard'}{('_' + args.tag) if args.tag else ''}.json" (out / name).write_text(json.dumps(R, ensure_ascii=False, indent=1), encoding="utf-8") print("->", out / name)