Add verified Verdict Core ML L512 FP16
Browse files- README.md +5 -5
- native_reference.py +11 -0
- preprocessing.py +4 -2
- verdict_fp16_L512_candidates25.conversion.json +15 -0
- verdict_fp16_L512_candidates25.mlpackage/Data/com.apple.CoreML/model.mlmodel +3 -0
- verdict_fp16_L512_candidates25.mlpackage/Data/com.apple.CoreML/weights/weight.bin +3 -0
- verdict_fp16_L512_candidates25.mlpackage/Manifest.json +18 -0
- verification-L512.json +216 -0
README.md
CHANGED
|
@@ -10,14 +10,14 @@ tags:
|
|
| 10 |
- verdict
|
| 11 |
---
|
| 12 |
|
| 13 |
-
# Verdict Core ML (L128 FP16)
|
| 14 |
|
| 15 |
-
This is a Core ML conversion of the **trained Verdict checkpoint** [`heman10x/rlcd-modernbert-151m`](https://huggingface.co/heman10x/rlcd-modernbert-151m), pinned at `8af2496eb63c7fa66d7d234e1f62629380030eb4`. It has 151,378,177 parameters and retains the full encoder plus trained decision head. The
|
| 16 |
|
| 17 |
-
Inputs are `input_ids` and `attention_mask` of shape `[1,
|
| 18 |
|
| 19 |
-
Verified on Apple M5 Pro/macOS 27.0 against the pinned PyTorch model:
|
| 20 |
|
| 21 |
-
The public Decision Index tracker reports Verdict at 13.38. Its historical checkpoint and renderer have not been authenticated against this pinned release, so this artifact does **not** claim to reproduce that score.
|
| 22 |
|
| 23 |
The Hub repo includes the runtime renderer, calibration helper, tokenizer, verification report and SHA-256 asset lock. Conversion source is maintained in [FluidInference/mobius](https://github.com/FluidInference/mobius). Upstream model and code: [Verdict model](https://huggingface.co/heman10x/rlcd-modernbert-151m), [Verdict source](https://github.com/Heman10x-NGU/Verdict-open-jev), Apache-2.0. Base architecture: [`knowledgator/gliclass-modern-base-v2.0`](https://huggingface.co/knowledgator/gliclass-modern-base-v2.0).
|
|
|
|
| 10 |
- verdict
|
| 11 |
---
|
| 12 |
|
| 13 |
+
# Verdict Core ML (L128 and L512 FP16)
|
| 14 |
|
| 15 |
+
This is a Core ML conversion of the **trained Verdict checkpoint** [`heman10x/rlcd-modernbert-151m`](https://huggingface.co/heman10x/rlcd-modernbert-151m), pinned at `8af2496eb63c7fa66d7d234e1f62629380030eb4`. It has 151,378,177 parameters and retains the full encoder plus trained decision head. The L128 package is 303,210,832 bytes; L512 is 304,390,482 bytes. Both target iOS 17 / macOS 14 or newer. Choose L128 when the full rendered request fits 128 tokens and L512 for 129–512 tokens. Do not truncate an overlength request into an installed bucket.
|
| 16 |
|
| 17 |
+
Inputs are `input_ids` and `attention_mask` of shape `[1,L]`, and `class_marker_map` of shape `[1,25,L]`, where `L` is 128 or 512. The graph returns raw logits and an uncalibrated softmax. For a native Verdict decision, render the request with `native_reference.py`, append the `__insufficient_evidence__` candidate, and apply `calibrator.json` to raw logits. This means at most **24 substantive candidates**. The released calibrator has a separate temperature for several candidate counts. Calling the graph with arbitrary labels or treating `probabilities` as calibrated changes the model's behavior.
|
| 18 |
|
| 19 |
+
Verified on Apple M5 Pro/macOS 27.0 against the pinned PyTorch model: L128 passed 4/4 smoke decisions (including two native abstentions), worst calibrated probability difference 0.00341. L512 passed 5/5 (including one 283-token public Decision Index row), worst difference 0.00070. The 20-iteration L128 `coreml-cli` median was 3.627 ms CPU+GPU, 3.710 ms CPU+ANE, 3.915 ms automatic, and 11.721 ms CPU only. L512 median automatic model call time across its five parity requests was 8.15 ms. Rendering and tokenization add time. The verification and profile JSON files include the details.
|
| 20 |
|
| 21 |
+
The public Decision Index tracker reports Verdict at 13.38. Its historical checkpoint and renderer have not been authenticated against this pinned release, so this artifact does **not** claim to reproduce that score. The two buckets cover its 512-token public limit, but full-suite correctness and quantized parity are still pending.
|
| 22 |
|
| 23 |
The Hub repo includes the runtime renderer, calibration helper, tokenizer, verification report and SHA-256 asset lock. Conversion source is maintained in [FluidInference/mobius](https://github.com/FluidInference/mobius). Upstream model and code: [Verdict model](https://huggingface.co/heman10x/rlcd-modernbert-151m), [Verdict source](https://github.com/Heman10x-NGU/Verdict-open-jev), Apache-2.0. Base architecture: [`knowledgator/gliclass-modern-base-v2.0`](https://huggingface.co/knowledgator/gliclass-modern-base-v2.0).
|
native_reference.py
CHANGED
|
@@ -62,6 +62,8 @@ def build_request(context: str, question: dict[str, Any]) -> Request:
|
|
| 62 |
raise ValueError(f"unknown Verdict question type: {kind}")
|
| 63 |
labels += (ABSTAIN_DESCRIPTION,)
|
| 64 |
ids += (ABSTAIN_ID,)
|
|
|
|
|
|
|
| 65 |
prefix = "".join(f"<<LABEL>>{label}" for label in labels)
|
| 66 |
return Request(kind, f"{prefix}<<SEP>>{text}", labels, ids, values)
|
| 67 |
|
|
@@ -86,10 +88,14 @@ def decode(logits: np.ndarray, request: Request, calibrator: dict[str, Any]) ->
|
|
| 86 |
distribution = dict(zip(request.ids, map(float, p)))
|
| 87 |
selected_id = request.ids[int(p.argmax())]
|
| 88 |
abstained = selected_id == ABSTAIN_ID
|
|
|
|
|
|
|
| 89 |
result: dict[str, Any] = {
|
| 90 |
"type": request.kind,
|
| 91 |
"selected_id": selected_id,
|
|
|
|
| 92 |
"probabilities": distribution,
|
|
|
|
| 93 |
"is_abstention": abstained,
|
| 94 |
"p_abstain": distribution[ABSTAIN_ID],
|
| 95 |
"calibration_status": "calibrated_for_scope",
|
|
@@ -99,6 +105,7 @@ def decode(logits: np.ndarray, request: Request, calibrator: dict[str, Any]) ->
|
|
| 99 |
elif request.kind == "noul":
|
| 100 |
substantive = distribution["true"] + distribution["false"]
|
| 101 |
result["noul"] = distribution["true"] / substantive if substantive and not abstained else None
|
|
|
|
| 102 |
else:
|
| 103 |
substantive = sum(distribution[key] for key in request.ids[:-1])
|
| 104 |
result["score"] = (
|
|
@@ -106,6 +113,10 @@ def decode(logits: np.ndarray, request: Request, calibrator: dict[str, Any]) ->
|
|
| 106 |
if substantive and not abstained
|
| 107 |
else None
|
| 108 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 109 |
return result
|
| 110 |
|
| 111 |
|
|
|
|
| 62 |
raise ValueError(f"unknown Verdict question type: {kind}")
|
| 63 |
labels += (ABSTAIN_DESCRIPTION,)
|
| 64 |
ids += (ABSTAIN_ID,)
|
| 65 |
+
if len(ids) != len(set(ids)):
|
| 66 |
+
raise ValueError("Verdict candidate IDs must be unique and cannot use the abstention ID")
|
| 67 |
prefix = "".join(f"<<LABEL>>{label}" for label in labels)
|
| 68 |
return Request(kind, f"{prefix}<<SEP>>{text}", labels, ids, values)
|
| 69 |
|
|
|
|
| 88 |
distribution = dict(zip(request.ids, map(float, p)))
|
| 89 |
selected_id = request.ids[int(p.argmax())]
|
| 90 |
abstained = selected_id == ABSTAIN_ID
|
| 91 |
+
entropy = -sum(float(value) * math.log(float(value)) for value in p if value > 1e-12)
|
| 92 |
+
concentration = max(0.0, min(1.0, 1.0 - entropy / math.log(k))) if k > 1 else 1.0
|
| 93 |
result: dict[str, Any] = {
|
| 94 |
"type": request.kind,
|
| 95 |
"selected_id": selected_id,
|
| 96 |
+
"selected_probability": distribution[selected_id],
|
| 97 |
"probabilities": distribution,
|
| 98 |
+
"concentration": concentration,
|
| 99 |
"is_abstention": abstained,
|
| 100 |
"p_abstain": distribution[ABSTAIN_ID],
|
| 101 |
"calibration_status": "calibrated_for_scope",
|
|
|
|
| 105 |
elif request.kind == "noul":
|
| 106 |
substantive = distribution["true"] + distribution["false"]
|
| 107 |
result["noul"] = distribution["true"] / substantive if substantive and not abstained else None
|
| 108 |
+
result["selected_outcome"] = selected_id
|
| 109 |
else:
|
| 110 |
substantive = sum(distribution[key] for key in request.ids[:-1])
|
| 111 |
result["score"] = (
|
|
|
|
| 113 |
if substantive and not abstained
|
| 114 |
else None
|
| 115 |
)
|
| 116 |
+
result["selected_level_id"] = selected_id
|
| 117 |
+
result["selected_value"] = (
|
| 118 |
+
request.values[request.ids.index(selected_id)] if not abstained else None
|
| 119 |
+
)
|
| 120 |
return result
|
| 121 |
|
| 122 |
|
preprocessing.py
CHANGED
|
@@ -5,13 +5,15 @@ from __future__ import annotations
|
|
| 5 |
import numpy as np
|
| 6 |
|
| 7 |
|
| 8 |
-
def prepare(
|
|
|
|
|
|
|
| 9 |
full = tokenizer(rendered, truncation=False)
|
| 10 |
if len(full["input_ids"]) > length:
|
| 11 |
raise ValueError(f"Verdict prompt needs {len(full['input_ids'])} tokens; L{length} has no room")
|
| 12 |
encoded = tokenizer(rendered, truncation=False, padding="max_length", max_length=length, return_tensors="np")
|
| 13 |
ids = encoded["input_ids"].astype(np.int32)
|
| 14 |
-
positions = np.flatnonzero(ids[0] ==
|
| 15 |
if len(positions) > max_candidates:
|
| 16 |
raise ValueError("candidate markers exceed exported head capacity")
|
| 17 |
markers = np.zeros((1, max_candidates, length), dtype=np.float32)
|
|
|
|
| 5 |
import numpy as np
|
| 6 |
|
| 7 |
|
| 8 |
+
def prepare(
|
| 9 |
+
tokenizer, class_token_index: int, rendered: str, length: int, max_candidates: int
|
| 10 |
+
) -> dict[str, np.ndarray]:
|
| 11 |
full = tokenizer(rendered, truncation=False)
|
| 12 |
if len(full["input_ids"]) > length:
|
| 13 |
raise ValueError(f"Verdict prompt needs {len(full['input_ids'])} tokens; L{length} has no room")
|
| 14 |
encoded = tokenizer(rendered, truncation=False, padding="max_length", max_length=length, return_tensors="np")
|
| 15 |
ids = encoded["input_ids"].astype(np.int32)
|
| 16 |
+
positions = np.flatnonzero(ids[0] == class_token_index)
|
| 17 |
if len(positions) > max_candidates:
|
| 18 |
raise ValueError("candidate markers exceed exported head capacity")
|
| 19 |
markers = np.zeros((1, max_candidates, length), dtype=np.float32)
|
verdict_fp16_L512_candidates25.conversion.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"package": "verdict_fp16_L512_candidates25.mlpackage",
|
| 3 |
+
"source_repo": "heman10x/rlcd-modernbert-151m",
|
| 4 |
+
"source_revision": "8af2496eb63c7fa66d7d234e1f62629380030eb4",
|
| 5 |
+
"source_weight_sha256": "d252823994d47a7933217fc86449493299643af6a0c0d83d6bd5a7666d3253ef",
|
| 6 |
+
"parameters": 151378177,
|
| 7 |
+
"wrapper_max_logit_error": 1.9073486328125e-06,
|
| 8 |
+
"conversion_seconds": 5.1714104160200804,
|
| 9 |
+
"package_bytes": 304390482,
|
| 10 |
+
"package_files_sha256": {
|
| 11 |
+
"Manifest.json": "047cc80f7d66d822ca0d9c0100a6a6dc96ae2d70eac42e292a2b246e01e64ddd",
|
| 12 |
+
"Data/com.apple.CoreML/model.mlmodel": "7ad21a9ee93d1c573e6d89e6c35511f08c984a5c47108bb2b908c96344c03903",
|
| 13 |
+
"Data/com.apple.CoreML/weights/weight.bin": "fb438de5823dd91c8448b373d026b7e6dcb7149bf871fde54057fd6d82586886"
|
| 14 |
+
}
|
| 15 |
+
}
|
verdict_fp16_L512_candidates25.mlpackage/Data/com.apple.CoreML/model.mlmodel
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7ad21a9ee93d1c573e6d89e6c35511f08c984a5c47108bb2b908c96344c03903
|
| 3 |
+
size 306985
|
verdict_fp16_L512_candidates25.mlpackage/Data/com.apple.CoreML/weights/weight.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fb438de5823dd91c8448b373d026b7e6dcb7149bf871fde54057fd6d82586886
|
| 3 |
+
size 304082880
|
verdict_fp16_L512_candidates25.mlpackage/Manifest.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"fileFormatVersion": "1.0.0",
|
| 3 |
+
"itemInfoEntries": {
|
| 4 |
+
"8DCAA6D5-D8F6-4FC5-B35A-D8D1776F73F2": {
|
| 5 |
+
"author": "com.apple.CoreML",
|
| 6 |
+
"description": "CoreML Model Specification",
|
| 7 |
+
"name": "model.mlmodel",
|
| 8 |
+
"path": "com.apple.CoreML/model.mlmodel"
|
| 9 |
+
},
|
| 10 |
+
"9CE3A02A-DD4B-448B-AA54-2D34CD87C941": {
|
| 11 |
+
"author": "com.apple.CoreML",
|
| 12 |
+
"description": "CoreML Model Weights",
|
| 13 |
+
"name": "weights",
|
| 14 |
+
"path": "com.apple.CoreML/weights"
|
| 15 |
+
}
|
| 16 |
+
},
|
| 17 |
+
"rootModelIdentifier": "8DCAA6D5-D8F6-4FC5-B35A-D8D1776F73F2"
|
| 18 |
+
}
|
verification-L512.json
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"source_repo": "heman10x/rlcd-modernbert-151m",
|
| 3 |
+
"source_revision": "8af2496eb63c7fa66d7d234e1f62629380030eb4",
|
| 4 |
+
"package": "verdict_fp16_L512_candidates25.mlpackage",
|
| 5 |
+
"chip": "arm",
|
| 6 |
+
"macos": "27.0",
|
| 7 |
+
"max_calibrated_probability_error": 0.0006975581321209678,
|
| 8 |
+
"selected_id_agreements": 5,
|
| 9 |
+
"questions": 5,
|
| 10 |
+
"rows": [
|
| 11 |
+
{
|
| 12 |
+
"name": "lost card",
|
| 13 |
+
"type": "choice",
|
| 14 |
+
"tokens": 53,
|
| 15 |
+
"candidates_including_abstention": 3,
|
| 16 |
+
"native": {
|
| 17 |
+
"type": "choice",
|
| 18 |
+
"selected_id": "card_lost",
|
| 19 |
+
"selected_probability": 0.6279954438939314,
|
| 20 |
+
"probabilities": {
|
| 21 |
+
"card_lost": 0.6279954438939314,
|
| 22 |
+
"pin_reset": 0.12699384187218735,
|
| 23 |
+
"__insufficient_evidence__": 0.2450107142338813
|
| 24 |
+
},
|
| 25 |
+
"concentration": 0.18185852825435922,
|
| 26 |
+
"is_abstention": false,
|
| 27 |
+
"p_abstain": 0.2450107142338813,
|
| 28 |
+
"calibration_status": "calibrated_for_scope",
|
| 29 |
+
"choice": "card_lost"
|
| 30 |
+
},
|
| 31 |
+
"coreml": {
|
| 32 |
+
"type": "choice",
|
| 33 |
+
"selected_id": "card_lost",
|
| 34 |
+
"selected_probability": 0.6286024758124755,
|
| 35 |
+
"probabilities": {
|
| 36 |
+
"card_lost": 0.6286024758124755,
|
| 37 |
+
"pin_reset": 0.12674703377209953,
|
| 38 |
+
"__insufficient_evidence__": 0.24465049041542497
|
| 39 |
+
},
|
| 40 |
+
"concentration": 0.18252696123986079,
|
| 41 |
+
"is_abstention": false,
|
| 42 |
+
"p_abstain": 0.24465049041542497,
|
| 43 |
+
"calibration_status": "calibrated_for_scope",
|
| 44 |
+
"choice": "card_lost"
|
| 45 |
+
},
|
| 46 |
+
"max_calibrated_probability_error": 0.0006070319185441653,
|
| 47 |
+
"selected_id_agrees": true,
|
| 48 |
+
"median_model_call_ms": 8.042125002248213
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"name": "binary inquiry",
|
| 52 |
+
"type": "noul",
|
| 53 |
+
"tokens": 50,
|
| 54 |
+
"candidates_including_abstention": 3,
|
| 55 |
+
"native": {
|
| 56 |
+
"type": "noul",
|
| 57 |
+
"selected_id": "true",
|
| 58 |
+
"selected_probability": 0.4692844463560151,
|
| 59 |
+
"probabilities": {
|
| 60 |
+
"true": 0.4692844463560151,
|
| 61 |
+
"false": 0.24040823180168494,
|
| 62 |
+
"__insufficient_evidence__": 0.29030732184229996
|
| 63 |
+
},
|
| 64 |
+
"concentration": 0.03808302501474403,
|
| 65 |
+
"is_abstention": false,
|
| 66 |
+
"p_abstain": 0.29030732184229996,
|
| 67 |
+
"calibration_status": "calibrated_for_scope",
|
| 68 |
+
"noul": 0.6612502295701238,
|
| 69 |
+
"selected_outcome": "true"
|
| 70 |
+
},
|
| 71 |
+
"coreml": {
|
| 72 |
+
"type": "noul",
|
| 73 |
+
"selected_id": "true",
|
| 74 |
+
"selected_probability": 0.46998200448813604,
|
| 75 |
+
"probabilities": {
|
| 76 |
+
"true": 0.46998200448813604,
|
| 77 |
+
"false": 0.23993786932412314,
|
| 78 |
+
"__insufficient_evidence__": 0.29008012618774076
|
| 79 |
+
},
|
| 80 |
+
"concentration": 0.03846968937399986,
|
| 81 |
+
"is_abstention": false,
|
| 82 |
+
"p_abstain": 0.29008012618774076,
|
| 83 |
+
"calibration_status": "calibrated_for_scope",
|
| 84 |
+
"noul": 0.6620211967927305,
|
| 85 |
+
"selected_outcome": "true"
|
| 86 |
+
},
|
| 87 |
+
"max_calibrated_probability_error": 0.0006975581321209678,
|
| 88 |
+
"selected_id_agrees": true,
|
| 89 |
+
"median_model_call_ms": 8.013729006052017
|
| 90 |
+
},
|
| 91 |
+
{
|
| 92 |
+
"name": "urgency score",
|
| 93 |
+
"type": "score",
|
| 94 |
+
"tokens": 55,
|
| 95 |
+
"candidates_including_abstention": 4,
|
| 96 |
+
"native": {
|
| 97 |
+
"type": "score",
|
| 98 |
+
"selected_id": "__insufficient_evidence__",
|
| 99 |
+
"selected_probability": 0.3259148157325998,
|
| 100 |
+
"probabilities": {
|
| 101 |
+
"low": 0.19793214504236917,
|
| 102 |
+
"medium": 0.2898236019444062,
|
| 103 |
+
"high": 0.18632943728062487,
|
| 104 |
+
"__insufficient_evidence__": 0.3259148157325998
|
| 105 |
+
},
|
| 106 |
+
"concentration": 0.02039165010923949,
|
| 107 |
+
"is_abstention": true,
|
| 108 |
+
"p_abstain": 0.3259148157325998,
|
| 109 |
+
"calibration_status": "calibrated_for_scope",
|
| 110 |
+
"score": null,
|
| 111 |
+
"selected_level_id": "__insufficient_evidence__",
|
| 112 |
+
"selected_value": null
|
| 113 |
+
},
|
| 114 |
+
"coreml": {
|
| 115 |
+
"type": "score",
|
| 116 |
+
"selected_id": "__insufficient_evidence__",
|
| 117 |
+
"selected_probability": 0.3262236353210236,
|
| 118 |
+
"probabilities": {
|
| 119 |
+
"low": 0.19780841262370594,
|
| 120 |
+
"medium": 0.28950115689828004,
|
| 121 |
+
"high": 0.18646679515699038,
|
| 122 |
+
"__insufficient_evidence__": 0.3262236353210236
|
| 123 |
+
},
|
| 124 |
+
"concentration": 0.020408360566200767,
|
| 125 |
+
"is_abstention": true,
|
| 126 |
+
"p_abstain": 0.3262236353210236,
|
| 127 |
+
"calibration_status": "calibrated_for_scope",
|
| 128 |
+
"score": null,
|
| 129 |
+
"selected_level_id": "__insufficient_evidence__",
|
| 130 |
+
"selected_value": null
|
| 131 |
+
},
|
| 132 |
+
"max_calibrated_probability_error": 0.000322445046126163,
|
| 133 |
+
"selected_id_agrees": true,
|
| 134 |
+
"median_model_call_ms": 8.470937464153394
|
| 135 |
+
},
|
| 136 |
+
{
|
| 137 |
+
"name": "insufficient evidence",
|
| 138 |
+
"type": "choice",
|
| 139 |
+
"tokens": 41,
|
| 140 |
+
"candidates_including_abstention": 3,
|
| 141 |
+
"native": {
|
| 142 |
+
"type": "choice",
|
| 143 |
+
"selected_id": "__insufficient_evidence__",
|
| 144 |
+
"selected_probability": 0.6003396635578521,
|
| 145 |
+
"probabilities": {
|
| 146 |
+
"visa": 0.19581437244279193,
|
| 147 |
+
"mastercard": 0.2038459639993559,
|
| 148 |
+
"__insufficient_evidence__": 0.6003396635578521
|
| 149 |
+
},
|
| 150 |
+
"concentration": 0.1354398243565177,
|
| 151 |
+
"is_abstention": true,
|
| 152 |
+
"p_abstain": 0.6003396635578521,
|
| 153 |
+
"calibration_status": "calibrated_for_scope",
|
| 154 |
+
"choice": "__insufficient_evidence__"
|
| 155 |
+
},
|
| 156 |
+
"coreml": {
|
| 157 |
+
"type": "choice",
|
| 158 |
+
"selected_id": "__insufficient_evidence__",
|
| 159 |
+
"selected_probability": 0.6008594846026971,
|
| 160 |
+
"probabilities": {
|
| 161 |
+
"visa": 0.19567826624893642,
|
| 162 |
+
"mastercard": 0.2034622491483665,
|
| 163 |
+
"__insufficient_evidence__": 0.6008594846026971
|
| 164 |
+
},
|
| 165 |
+
"concentration": 0.13595645758378905,
|
| 166 |
+
"is_abstention": true,
|
| 167 |
+
"p_abstain": 0.6008594846026971,
|
| 168 |
+
"calibration_status": "calibrated_for_scope",
|
| 169 |
+
"choice": "__insufficient_evidence__"
|
| 170 |
+
},
|
| 171 |
+
"max_calibrated_probability_error": 0.000519821044845048,
|
| 172 |
+
"selected_id_agrees": true,
|
| 173 |
+
"median_model_call_ms": 8.155353978509083
|
| 174 |
+
},
|
| 175 |
+
{
|
| 176 |
+
"name": "BFCL-tool-selection:simple:simple_254",
|
| 177 |
+
"type": "choice",
|
| 178 |
+
"tokens": 283,
|
| 179 |
+
"candidates_including_abstention": 3,
|
| 180 |
+
"native": {
|
| 181 |
+
"type": "choice",
|
| 182 |
+
"selected_id": "yes",
|
| 183 |
+
"selected_probability": 0.5095718104586857,
|
| 184 |
+
"probabilities": {
|
| 185 |
+
"yes": 0.5095718104586857,
|
| 186 |
+
"no": 0.28447377613128705,
|
| 187 |
+
"__insufficient_evidence__": 0.20595441341002724
|
| 188 |
+
},
|
| 189 |
+
"concentration": 0.0655574261710149,
|
| 190 |
+
"is_abstention": false,
|
| 191 |
+
"p_abstain": 0.20595441341002724,
|
| 192 |
+
"calibration_status": "calibrated_for_scope",
|
| 193 |
+
"choice": "yes"
|
| 194 |
+
},
|
| 195 |
+
"coreml": {
|
| 196 |
+
"type": "choice",
|
| 197 |
+
"selected_id": "yes",
|
| 198 |
+
"selected_probability": 0.5093345700124705,
|
| 199 |
+
"probabilities": {
|
| 200 |
+
"yes": 0.5093345700124705,
|
| 201 |
+
"no": 0.28465867339056805,
|
| 202 |
+
"__insufficient_evidence__": 0.20600675659696147
|
| 203 |
+
},
|
| 204 |
+
"concentration": 0.06541626747983909,
|
| 205 |
+
"is_abstention": false,
|
| 206 |
+
"p_abstain": 0.20600675659696147,
|
| 207 |
+
"calibration_status": "calibrated_for_scope",
|
| 208 |
+
"choice": "yes"
|
| 209 |
+
},
|
| 210 |
+
"max_calibrated_probability_error": 0.00023724044621520335,
|
| 211 |
+
"selected_id_agrees": true,
|
| 212 |
+
"median_model_call_ms": 8.165458013536409
|
| 213 |
+
}
|
| 214 |
+
],
|
| 215 |
+
"passed": true
|
| 216 |
+
}
|