"""한국어 도구 호출 시험(Unsloth Studio, 생각 끔, temperature 0). 2026-10-11 5060 Ti, Saluki 27B 글에서 처음 씀. 모델에 도구 6개를 주고 한국어 부탁 20개를 보낸다. 모델이 **부른 도구의 이름과 인자**가 맞는지만 본다(도구를 실제로 실행하지는 않는다). - 도구 하나를 불러야 하는 부탁 12개 - 도구 두 개를 한 번에 불러야 하는 부탁 4개(병렬 호출) - 도구를 부르면 안 되는 말 4개 각 부탁을 --runs 번(기본 3) 보낸다. 채점은 느슨하게: 이름이 맞고, 꼭 있어야 하는 값이 인자에 들어 있으면 통과. 사용: python tool_call_test.py <모델 id> <이름표> [--variant X] [--runs 3] [--ctx 32768] 결과: results/<날짜>__tools_<이름표>.json """ import argparse, json, re, socket, statistics, sys, time from pathlib import Path import requests sys.stdout.reconfigure(encoding="utf-8") HERE = Path(__file__).resolve().parent sys.path.insert(0, str(HERE.parent)) from local_env import get # noqa: E402 API = get("UNSLOTH_URL", "http://127.0.0.1:8888") H = {"Authorization": "Bearer " + get("UNSLOTH_API_KEY")} PC = {"woojungho_28590": "5090", "woojungho_1660": "5060ti"}.get(socket.gethostname().lower(), socket.gethostname()) ap = argparse.ArgumentParser() ap.add_argument("ref"); ap.add_argument("tag"); ap.add_argument("--variant") ap.add_argument("--runs", type=int, default=3); ap.add_argument("--ctx", type=int, default=32768) args = ap.parse_args() def fn(name, desc, props, req): return {"type": "function", "function": {"name": name, "description": desc, "parameters": {"type": "object", "properties": props, "required": req}}} S = lambda d: {"type": "string", "description": d} TOOLS = [ fn("get_weather", "도시의 날씨를 알려 준다", {"city": S("도시 이름"), "date": S("날짜. 오늘이면 비워 둔다")}, ["city"]), fn("convert_currency", "돈을 다른 나라 돈으로 바꿔 계산한다", {"amount": {"type": "number", "description": "금액"}, "from": S("원래 통화 코드. 예: USD"), "to": S("바꿀 통화 코드. 예: KRW")}, ["amount", "from", "to"]), fn("search_web", "인터넷에서 검색한다", {"query": S("검색어")}, ["query"]), fn("set_alarm", "알람을 맞춘다", {"time": S("24시간 형식 HH:MM"), "label": S("알람 이름")}, ["time"]), fn("calculate", "수식을 계산한다", {"expression": S("계산할 수식. 예: 12*3+4")}, ["expression"]), fn("send_message", "사람에게 문자를 보낸다", {"to": S("받는 사람"), "text": S("보낼 내용")}, ["to", "text"]), ] # (id, 부탁, [기대하는 호출: (도구 이름, {인자: 들어 있어야 하는 값(정규식)})]) 빈 목록이면 도구를 부르면 안 된다 CASES = [ ("s01", "내일 부산 날씨 알려 줘", [("get_weather", {"city": "부산|busan"})]), ("s02", "150달러가 원화로 얼마야?", [("convert_currency", {"amount": "^150(\\.0)?$", "from": "usd", "to": "krw"})]), ("s03", "아침 7시 30분에 알람 맞춰 줘", [("set_alarm", {"time": "^0?7:30"})]), ("s04", "37만 8천 원의 15%가 얼마인지 계산해 줘", [("calculate", {"expression": "378[,]?000"})]), ("s05", "엄마한테 '오늘 늦어요'라고 문자 보내 줘", [("send_message", {"to": "엄마", "text": "늦"})]), ("s06", "요즘 RTX 5060 Ti 가격 좀 찾아봐 줘", [("search_web", {"query": "5060"})]), ("s07", "도쿄 지금 비 와?", [("get_weather", {"city": "도쿄|tokyo|東京"})]), ("s08", "20만 엔은 한국 돈으로 얼마야?", [("convert_currency", {"amount": "^200[,]?000(\\.0)?$", "from": "jpy", "to": "krw"})]), ("s09", "오후 3시에 회의 알람 맞춰 줘", [("set_alarm", {"time": "^15:00"})]), ("s10", "1234 곱하기 5678을 계산해 줘", [("calculate", {"expression": "1234.*5678"})]), ("s11", "김 대리에게 회의 자료 보냈다고 문자로 알려 줘", [("send_message", {"to": "김", "text": "자료"})]), ("s12", "파이썬 3.15에서 뭐가 바뀌었는지 검색해 줘", [("search_web", {"query": "3\\.15"})]), ("p01", "서울이랑 제주 날씨 둘 다 알려 줘", [("get_weather", {"city": "서울|seoul"}), ("get_weather", {"city": "제주|jeju"})]), ("p02", "100달러와 100유로가 각각 원화로 얼마인지 알려 줘", [("convert_currency", {"amount": "^100(\\.0)?$", "from": "usd", "to": "krw"}), ("convert_currency", {"amount": "^100(\\.0)?$", "from": "eur", "to": "krw"})]), ("p03", "아침 6시와 6시 10분에 알람을 하나씩 맞춰 줘", [("set_alarm", {"time": "^0?6:00"}), ("set_alarm", {"time": "^0?6:10"})]), ("p04", "부산 날씨 알려 주고, 저녁 7시에 알람도 맞춰 줘", [("get_weather", {"city": "부산|busan"}), ("set_alarm", {"time": "^19:00"})]), ("n01", "안녕, 반가워!", []), ("n02", "고마워, 도움이 됐어.", []), ("n03", "'사과'를 영어로 하면 뭐야?", []), ("n04", "너는 어떤 도구를 쓸 수 있어?", []), ] def api(method, path, **kw): kw.setdefault("timeout", 900) return requests.request(method, API + path, headers=H, **kw) def loaded(): return api("GET", "/api/inference/loaded-models").json().get("data", []) def match(call, want): name, need = want if call["name"] != name: return False a = call["args"] if isinstance(call["args"], dict) else {} return all(re.search(pat, str(a.get(k, "")), re.I) for k, pat in need.items()) def grade(calls, wants): if not wants: return len(calls) == 0 if len(calls) != len(wants): return False left = list(calls) for w in wants: hit = next((c for c in left if match(c, w)), None) if hit is None: return False left.remove(hit) return True if loaded(): sys.exit("이미 올라간 모델이 있습니다. 내린 뒤 다시 실행하세요.") body = {"model_path": args.ref, "max_seq_length": args.ctx, "force_reload": True} if args.variant: body["gguf_variant"] = args.variant r = api("POST", "/v1/load", json=body) lm = loaded() if r.status_code != 200 or not lm: sys.exit(f"로드 실패 {r.status_code} {r.text[:300]}") mid = lm[0]["id"] R = {"started": time.strftime("%Y-%m-%d %H:%M:%S"), "pc": PC, "model": mid, "variant": args.variant, "ctx": lm[0].get("context_length"), "runs": args.runs, "setting": "enable_thinking false, temperature 0, tools 6개", "tools": [t["function"]["name"] for t in TOOLS], "rows": []} try: for cid, text, wants in CASES: row = {"id": cid, "text": text, "want": [[w[0], w[1]] for w in wants], "tries": []} for _ in range(args.runs): t = time.perf_counter() resp = api("POST", "/v1/chat/completions", json={"model": mid, "messages": [{"role": "user", "content": text}], "tools": TOOLS, "tool_choice": "auto", "temperature": 0, "enable_thinking": False, "max_tokens": 600, "stream": False}) dt = time.perf_counter() - t try: msg = resp.json()["choices"][0]["message"] except Exception: row["tries"].append({"ok": False, "error": f"HTTP {resp.status_code} {resp.text[:200]}", "sec": round(dt, 2)}); continue calls = [] for c in msg.get("tool_calls") or []: raw = c["function"].get("arguments") try: a = json.loads(raw) if isinstance(raw, str) else raw except Exception: a = {"_raw": raw} calls.append({"name": c["function"]["name"], "args": a}) row["tries"].append({"ok": grade(calls, wants), "calls": calls, "content": (msg.get("content") or "")[:200], "sec": round(dt, 2)}) row["pass"] = sum(x["ok"] for x in row["tries"]) R["rows"].append(row) print(cid, f'{row["pass"]}/{args.runs}', text, "|", json.dumps(row["tries"][0].get("calls"), ensure_ascii=False)[:150], flush=True) finally: for m in loaded(): api("POST", "/v1/unload", json={"model_path": m["id"]}, timeout=180) grp = lambda p: [r for r in R["rows"] if r["id"].startswith(p)] R["summary"] = {k: {"pass_tries": sum(r["pass"] for r in grp(p)), "tries": len(grp(p)) * args.runs, "all_pass_cases": sum(r["pass"] == args.runs for r in grp(p)), "cases": len(grp(p))} for k, p in (("single", "s"), ("parallel", "p"), ("no_tool", "n"))} R["sec_mean"] = round(statistics.mean(x["sec"] for r in R["rows"] for x in r["tries"]), 2) R["finished"] = time.strftime("%Y-%m-%d %H:%M:%S") out = HERE / "results" / f"{time.strftime('%Y-%m-%d')}_{PC}_tools_{args.tag}.json" out.write_text(json.dumps(R, ensure_ascii=False, indent=1), encoding="utf-8") print(R["summary"], "평균", R["sec_mean"], "초 ->", out.name)