{ "scope": "Fixed local regression cases, not general multimodal accuracy or calibration proof. One measured run per path per case.", "cases": [ { "name": "skunks", "expected": { "Baby skunks": 5, "Dog": 1 }, "result": { "answers": { "Baby skunks": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 14.98673152923584, "1": 17.359764099121094, "2": 16.392131805419922, "3": 19.846782684326172, "4": 20.9791316986084, "5": 21.16490936279297, "6": 20.06216049194336, "7": 18.140613555908203, "8": 16.926198959350586, "9": 16.051298141479492 }, "allowed_token_mass": 0.9999961256980896, "input_tokens": 713 }, "confidence": 0.3821704513596691, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 0.0008192375177709055, "1": 0.008790321988016542, "2": 0.0033401611305301847, "3": 0.10570687437322064, "4": 0.32800174932805193, "5": 0.39496478375333416, "6": 0.13111145133159421, "7": 0.0191921395925517, "8": 0.005697834568684499, "9": 0.0023754464162451497 }, "type": "score", "score": 4.607397562880727, "legend": { "0": "0 baby skunks", "1": "1 baby skunks", "2": "2 baby skunks", "3": "3 baby skunks", "4": "4 baby skunks", "5": "5 baby skunks", "6": "6 baby skunks", "7": "7 baby skunks", "8": "8 baby skunks", "9": "9 baby skunks" } }, "Dog": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 16.807817459106445, "1": 24.9267635345459, "2": 17.953956604003906, "3": 17.375385284423828, "4": 17.24942398071289, "5": 15.531764030456543, "6": 14.995515823364258, "7": 15.121858596801758, "8": 17.92765235900879, "9": 20.62111473083496 }, "allowed_token_mass": 0.999992311000824, "input_tokens": 711 }, "confidence": 0.957409749394999, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 0.000292916934495495, "1": 0.9834628513850748, "2": 0.0009215234401209153, "3": 0.0005166971368865352, "4": 0.00045554549474873527, "5": 8.176388481262041e-05, "6": 4.7826861277107867e-05, "7": 5.4267754268551914e-05, "8": 0.0008975994919063731, "9": 0.013269007616408781 }, "type": "score", "score": 1.116355691009507, "legend": { "0": "0 dogs", "1": "1 dogs", "2": "2 dogs", "3": "3 dogs", "4": "4 dogs", "5": "5 dogs", "6": "6 dogs", "7": "7 dogs", "8": "8 dogs", "9": "9 dogs" } } }, "inference_seconds": 1.4461837910057511, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 707, "question_suffix_tokens": [ 6, 4 ], "candidate_token_lengths": [ [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ] ], "preprocess_seconds": 0.28576045800582506, "prefill_seconds": 1.1155447500059381, "branch_seconds": 0.044699791993480176, "branch_batch_sizes": [ 2 ], "candidate_batch_sizes": [], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 1.4461837910057511 }, "comparison": { "seconds": { "parallel": 1.4461837910057511, "normal": 1.353163209016202 }, "normal_over_parallel": 0.9356785890091758, "parallel_values": { "Baby skunks": 5, "Dog": 1 }, "normal": { "answers": { "Baby skunks": 5, "Dog": 1 }, "raw_text": "{\"Baby skunks\": 5, \"Dog\": 1}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 707, "output_tokens": 15, "max_output_tokens": 128, "preprocess_seconds": 0.1666350000014063, "generation_seconds": 1.186489417013945, "inference_seconds": 1.353163209016202, "total_seconds": 1.353163209016202 }, "agreement": { "Baby skunks": true, "Dog": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "text", "expected": { "animal": "cat", "count": 2, "dog": false, "presence": { "cat": true, "dog": false } }, "result": { "answers": { "animal": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "cat": -20.418933868408203, "dog": -39.288429260253906, "horse": -41.22490692138672 }, "allowed_token_mass": 1.3557190473655992e-09, "input_tokens": 382 }, "confidence": 0.9999998662687544, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "cat": 0.9999999926955352, "dog": 6.383844089613385e-09, "horse": 9.206206440074525e-10 }, "type": "choice", "choice": "cat", "selected_probability": 0.9999999926955352 }, "count": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 12.217308044433594, "1": 8.934037208557129, "2": 21.711172103881836, "3": 6.471416473388672 }, "allowed_token_mass": 1.0, "input_tokens": 381 }, "confidence": 0.9993990589423729, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 7.530662585853558e-05, "1": 2.8244037711623705e-06, "2": 0.9999216282991354, "3": 2.406712349138733e-07 }, "type": "score", "score": 1.9998468030157468, "legend": { "0": "Zero", "1": "One", "2": "Two", "3": "Three" } }, "dog": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": -4.55612850189209, "false": 9.909834861755371 }, "allowed_token_mass": 0.9999980330467224, "input_tokens": 380 }, "type": "noul", "noul": 5.218091726916857e-07 }, "presence": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "type": "independent", "probabilities": { "cat": 0.9999998752336507, "dog": 2.0655771827521545e-09 }, "diagnostics": { "cat": { "logits": { "yes": 4.4790449142456055, "no": -11.417778015136719 }, "allowed_token_mass": 0.9999871253967285, "input_tokens": 383 }, "dog": { "logits": { "yes": -8.802624702453613, "no": 11.195231437683105 }, "allowed_token_mass": 0.9999980330467224, "input_tokens": 383 } } } }, "inference_seconds": 0.27812479200656526, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 377, "question_suffix_tokens": [ 4, 4, 3, 6, 6 ], "candidate_token_lengths": [ [ 2, 2, 2 ], [ 1, 1, 1, 1 ], [ 1, 1 ], [ 1, 1 ], [ 1, 1 ] ], "preprocess_seconds": 0.004685749998316169, "prefill_seconds": 0.12183699998422526, "branch_seconds": 0.15141454202239402, "branch_batch_sizes": [ 4 ], "candidate_batch_sizes": [ 3 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 0.27812479200656526 }, "comparison": { "seconds": { "parallel": 0.27812479200656526, "normal": 0.8187501250067726 }, "normal_over_parallel": 2.9438228756947553, "parallel_values": { "animal": "cat", "count": 2, "dog": false, "presence": { "cat": true, "dog": false } }, "normal": { "answers": { "animal": "cat", "count": 2, "dog": false, "presence": { "cat": true, "dog": false } }, "raw_text": "{\"animal\": \"cat\", \"count\": 2, \"dog\": false, \"presence\": {\"cat\": true, \"dog\": false}}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 377, "output_tokens": 31, "max_output_tokens": 128, "preprocess_seconds": 0.0008309170079883188, "generation_seconds": 0.817888874997152, "inference_seconds": 0.8187501250067726, "total_seconds": 0.8187501250067726 }, "agreement": { "animal": true, "count": true, "dog": true, "presence": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "image", "expected": { "document": "Invoice", "currency": "USD", "paid": false }, "result": { "answers": { "document": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "Invoice": -20.7464599609375, "Resume": -50.827484130859375, "Report": -54.91582489013672 }, "allowed_token_mass": 9.770727920921203e-10, "input_tokens": 562 }, "confidence": 0.9999999999975123, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "Invoice": 0.9999999999999123, "Resume": 8.629332295972805e-14, "Report": 1.4468828238803991e-15 }, "type": "choice", "choice": "Invoice", "selected_probability": 0.9999999999999123 }, "currency": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "USD": -19.25501251220703, "EUR": -49.80154037475586, "GBP": -52.17305374145508 }, "allowed_token_mass": 4.3416450702636266e-09, "input_tokens": 562 }, "confidence": 0.9999999999982881, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "USD": 0.9999999999999407, "EUR": 5.41765702818478e-14, "GBP": 5.056806540907407e-15 }, "type": "choice", "choice": "USD", "selected_probability": 0.9999999999999407 }, "paid": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": -8.800516128540039, "false": 5.705474853515625 }, "allowed_token_mass": 0.9989376664161682, "input_tokens": 560 }, "type": "noul", "noul": 5.013349063764397e-07 } }, "inference_seconds": 1.0797851250099484, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 557, "question_suffix_tokens": [ 4, 4, 3 ], "candidate_token_lengths": [ [ 2, 2, 2 ], [ 2, 2, 2 ], [ 1, 1 ] ], "preprocess_seconds": 0.023936167010106146, "prefill_seconds": 0.9365400829992723, "branch_seconds": 0.11906045800424181, "branch_batch_sizes": [ 1 ], "candidate_batch_sizes": [ 6 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 1.0797851250099484 }, "comparison": { "seconds": { "parallel": 1.0797851250099484, "normal": 1.115999042027397 }, "normal_over_parallel": 1.0335380773254448, "parallel_values": { "document": "Invoice", "currency": "USD", "paid": false }, "normal": { "answers": { "document": "Invoice", "currency": "USD", "paid": false }, "raw_text": "{\"document\": \"Invoice\", \"currency\": \"USD\", \"paid\": false}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 557, "output_tokens": 18, "max_output_tokens": 128, "preprocess_seconds": 0.013549916999181733, "generation_seconds": 1.1024205420108046, "inference_seconds": 1.115999042027397, "total_seconds": 1.115999042027397 }, "agreement": { "document": true, "currency": true, "paid": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "speech", "expected": { "intent": "Cancel", "timing": "End of this month", "refund": false }, "result": { "answers": { "intent": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "Cancel": -18.3853816986084, "Upgrade": -38.652488708496094, "Renew": -41.1723747253418 }, "allowed_token_mass": 1.0359294541292768e-08, "input_tokens": 424 }, "confidence": 0.9999999667034646, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "Cancel": 0.9999999982950192, "Upgrade": 1.5780009514239098e-09, "Renew": 1.2697980873624795e-10 }, "type": "choice", "choice": "Cancel", "selected_probability": 0.9999999982950192 }, "timing": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "End of this month": -15.572050094604492, "Immediately": -37.958675384521484, "Unspecified": -39.069862365722656 }, "allowed_token_mass": 1.7264125010526324e-07, "input_tokens": 427 }, "confidence": 0.9999999945750523, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "End of this month": 0.999999999748121, "Immediately": 1.895012888388315e-10, "Unspecified": 6.237776268029532e-11 }, "type": "choice", "choice": "End of this month", "selected_probability": 0.999999999748121 }, "refund": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": -1.6868846416473389, "false": 6.2722673416137695 }, "allowed_token_mass": 0.9913969039916992, "input_tokens": 422 }, "type": "noul", "noul": 0.00034932725855031644 } }, "inference_seconds": 1.3889339579909574, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 419, "question_suffix_tokens": [ 4, 4, 3 ], "candidate_token_lengths": [ [ 2, 2, 2 ], [ 5, 2, 3 ], [ 1, 1 ] ], "preprocess_seconds": 0.06908670798293315, "prefill_seconds": 1.1997922919981647, "branch_seconds": 0.11987662501633167, "branch_batch_sizes": [ 1 ], "candidate_batch_sizes": [ 6 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 1.3889339579909574 }, "comparison": { "seconds": { "parallel": 1.3889339579909574, "normal": 0.7224976250145119 }, "normal_over_parallel": 0.5201814102519161, "parallel_values": { "intent": "Cancel", "timing": "End of this month", "refund": false }, "normal": { "answers": { "intent": "Cancel", "timing": "End of this month", "refund": false }, "raw_text": "{\"intent\": \"Cancel\", \"timing\": \"End of this month\", \"refund\": false}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 419, "output_tokens": 21, "max_output_tokens": 128, "preprocess_seconds": 0.040587333001894876, "generation_seconds": 0.6818767499935348, "inference_seconds": 0.7224976250145119, "total_seconds": 0.7224976250145119 }, "agreement": { "intent": true, "timing": true, "refund": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "video", "expected": { "moving_object": "Red square", "direction": "Left to right", "blue_circle": true }, "result": { "answers": { "moving_object": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "Red square": -22.188453674316406, "Blue circle": -33.092933654785156 }, "allowed_token_mass": 2.3103883589986892e-10, "input_tokens": 616 }, "confidence": 0.9996844110131247, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "Red square": 0.9999816246112391, "Blue circle": 1.83753887610297e-05 }, "type": "choice", "choice": "Red square", "selected_probability": 0.9999816246112391 }, "direction": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "Left to right": -27.598617553710938, "Right to left": -28.00503158569336, "No movement": -16.263721466064453 }, "allowed_token_mass": 8.64498327487706e-08, "input_tokens": 615 }, "confidence": 0.9997735527900483, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "Left to right": 1.1948366371645229e-05, "Right to left": 7.95802243955188e-06, "No movement": 0.9999800936111888 }, "type": "choice", "choice": "No movement", "selected_probability": 0.9999800936111888 }, "blue_circle": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": 6.57560396194458, "false": 5.888490200042725 }, "allowed_token_mass": 0.9332365393638611, "input_tokens": 613 }, "type": "noul", "noul": 0.6653245614557216 } }, "inference_seconds": 1.0843171250016894, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 608, "question_suffix_tokens": [ 6, 4, 5 ], "candidate_token_lengths": [ [ 3, 3 ], [ 4, 4, 3 ], [ 1, 1 ] ], "preprocess_seconds": 0.07178533400292508, "prefill_seconds": 0.6009469580021687, "branch_seconds": 0.4114130829984788, "branch_batch_sizes": [ 1 ], "candidate_batch_sizes": [ 5 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 1.0843171250016894 }, "comparison": { "seconds": { "parallel": 1.0843171250016894, "normal": 1.3601389999967068 }, "normal_over_parallel": 1.2543738069198047, "parallel_values": { "moving_object": "Red square", "direction": "No movement", "blue_circle": true }, "normal": { "answers": { "moving_object": "Red square", "direction": "No movement", "blue_circle": true }, "raw_text": "{\"moving_object\": \"Red square\", \"direction\": \"No movement\", \"blue_circle\": true}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 608, "output_tokens": 24, "max_output_tokens": 128, "preprocess_seconds": 0.07225479101180099, "generation_seconds": 1.2878438339976128, "inference_seconds": 1.3601389999967068, "total_seconds": 1.3601389999967068 }, "agreement": { "moving_object": true, "direction": true, "blue_circle": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "video_speech", "expected": { "animal": "dog", "red": true, "blue": true }, "result": { "answers": { "animal": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "dog": -23.273460388183594, "cat": -40.067718505859375, "horse": -41.919883728027344 }, "allowed_token_mass": 7.806648118713585e-11, "input_tokens": 696 }, "confidence": 0.9999990335836774, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "dog": 0.99999994116428, "cat": 5.085648571784501e-08, "horse": 7.979234170655804e-09 }, "type": "choice", "choice": "dog", "selected_probability": 0.99999994116428 }, "red": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": -3.506460666656494, "false": -8.199481964111328 }, "allowed_token_mass": 0.001379593275487423, "input_tokens": 694 }, "type": "noul", "noul": 0.9909241530989752 }, "blue": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": 1.0378401279449463, "false": -5.780968189239502 }, "allowed_token_mass": 0.0015309207374230027, "input_tokens": 694 }, "type": "noul", "noul": 0.9989081707113475 } }, "inference_seconds": 0.8778675420035142, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 691, "question_suffix_tokens": [ 4, 3, 3 ], "candidate_token_lengths": [ [ 2, 2, 2 ], [ 1, 1 ], [ 1, 1 ] ], "preprocess_seconds": 0.0949398749799002, "prefill_seconds": 0.6934007500240114, "branch_seconds": 0.08924333399045281, "branch_batch_sizes": [ 2 ], "candidate_batch_sizes": [ 3 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 0.8778675420035142 }, "comparison": { "seconds": { "parallel": 0.8778675420035142, "normal": 1.2360690420027822 }, "normal_over_parallel": null, "parallel_values": { "animal": "dog", "red": true, "blue": true }, "normal": { "answers": null, "raw_text": "{\"animal\": \"dog\", \"red\": \"true\", \"blue\": \"true\"}", "valid": false, "error": "Generated value for red does not match its answer contract", "finish_reason": "stop", "input_tokens": 691, "output_tokens": 19, "max_output_tokens": 128, "preprocess_seconds": 0.0908876670000609, "generation_seconds": 1.1451280410110485, "inference_seconds": 1.2360690420027822, "total_seconds": 1.2360690420027822 }, "agreement": { "animal": null, "red": null, "blue": null }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "speech_skunks", "expected": { "skunks": 5, "dogs": 1, "call_vet": false, "action": "Do not call the vet" }, "result": { "answers": { "skunks": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 2.4332430362701416, "1": 0.33744150400161743, "2": 0.49974024295806885, "3": 1.0511709451675415, "4": 2.741050958633423, "5": 20.806188583374023, "6": 2.202913999557495, "7": -1.8478120565414429, "8": -2.2469234466552734, "9": 0.48146724700927734 }, "allowed_token_mass": 1.0, "input_tokens": 620 }, "confidence": 0.9999996565698799, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 1.0488928261791463e-08, "1": 1.2898406934682289e-09, "2": 1.5171255458984183e-09, "3": 2.6333272442131735e-09, "4": 1.4269553933267914e-08, "5": 0.9999999597381529, "6": 8.331064281204806e-09, "7": 1.4504157649251244e-10, "8": 9.731070908143261e-11, "9": 1.4896548671182722e-09 }, "type": "score", "score": 4.999999933180111, "legend": { "0": "0 baby skunks", "1": "1 baby skunks", "2": "2 baby skunks", "3": "3 baby skunks", "4": "4 baby skunks", "5": "5 baby skunks", "6": "6 baby skunks", "7": "7 baby skunks", "8": "8 baby skunks", "9": "9 baby skunks" } }, "dogs": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 12.110952377319336, "1": 25.3037166595459, "2": 9.88134765625, "3": 6.69423246383667, "4": 8.310088157653809, "5": 7.257328510284424, "6": 6.1623053550720215, "7": 4.915668487548828, "8": 5.334745407104492, "9": 7.838923931121826 }, "allowed_token_mass": 1.0, "input_tokens": 619 }, "confidence": 0.9999862804353773, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 1.86403615456672e-06, "1": 0.9999978365667048, "2": 2.005161255538126e-07, "3": 8.279474314556474e-09, "4": 4.16639052158653e-08, "5": 1.4539593643851128e-08, "6": 4.863957315537724e-09, "7": 1.3982416759546908e-09, "8": 2.126106602756648e-09, "9": 2.6009736406271668e-08 }, "type": "score", "score": 0.9999987918588841, "legend": { "0": "0 dogs", "1": "1 dogs", "2": "2 dogs", "3": "3 dogs", "4": "4 dogs", "5": "5 dogs", "6": "6 dogs", "7": "7 dogs", "8": "8 dogs", "9": "9 dogs" } }, "call_vet": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": -0.8286030292510986, "false": 6.238976955413818 }, "allowed_token_mass": 0.999083399772644, "input_tokens": 620 }, "type": "noul", "noul": 0.0008515673929433828 }, "action": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "Call the vet": -40.304771423339844, "Do not call the vet": -23.257143020629883 }, "allowed_token_mass": 7.93507687306582e-11, "input_tokens": 624 }, "confidence": 0.9999989722115856, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "Call the vet": 3.947380923962134e-08, "Do not call the vet": 0.9999999605261907 }, "type": "choice", "choice": "Do not call the vet", "selected_probability": 0.9999999605261907 } }, "inference_seconds": 0.57920274999924, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 615, "primitive_fields": 4, "question_suffix_tokens": [ 5, 4, 5, 4 ], "candidate_token_lengths": [ [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1 ], [ 4, 6 ] ], "preprocess_seconds": 0.10492087501916103, "prefill_seconds": 0.3842746669834014, "branch_seconds": 0.08970829099416733, "branch_batch_sizes": [ 3 ], "candidate_batch_sizes": [ 2 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 0.57920274999924 }, "comparison": { "seconds": { "parallel": 0.57920274999924, "normal": 0.7062484589987434 }, "normal_over_parallel": 1.2193458318346557, "parallel_values": { "skunks": 5, "dogs": 1, "call_vet": false, "action": "Do not call the vet" }, "normal": { "answers": { "skunks": 5, "dogs": 1, "call_vet": false, "action": "Do not call the vet" }, "raw_text": "{\"skunks\": 5, \"dogs\": 1, \"call_vet\": false, \"action\": \"Do not call the vet\"}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 615, "output_tokens": 31, "max_output_tokens": 128, "preprocess_seconds": 0.034028374997433275, "generation_seconds": 0.6721794169861823, "inference_seconds": 0.7062484589987434, "total_seconds": 0.7062484589987434 }, "agreement": { "skunks": true, "dogs": true, "call_vet": true, "action": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "speech_dogs", "expected": { "skunks": 0, "dogs": 2, "call_vet": true, "action": "Call the vet" }, "result": { "answers": { "skunks": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 27.120681762695312, "1": 16.34250259399414, "2": 17.907949447631836, "3": 11.980827331542969, "4": 10.985052108764648, "5": 11.182181358337402, "6": 11.026575088500977, "7": 7.479489326477051, "8": 8.592439651489258, "9": 13.021932601928711 }, "allowed_token_mass": 1.0, "input_tokens": 609 }, "confidence": 0.9994416628309183, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 0.9998780526402758, "1": 2.0846987105444333e-05, "2": 9.974892593119056e-05, "3": 2.6594498152027503e-07, "4": 9.824989953652488e-08, "5": 1.1965869449626406e-07, "6": 1.0241541188359713e-07, "7": 2.9504315445239566e-09, "8": 8.979119003119067e-09, "9": 7.532481495483551e-07 }, "type": "score", "score": 0.0002296201787730871, "legend": { "0": "0 baby skunks", "1": "1 baby skunks", "2": "2 baby skunks", "3": "3 baby skunks", "4": "4 baby skunks", "5": "5 baby skunks", "6": "6 baby skunks", "7": "7 baby skunks", "8": "8 baby skunks", "9": "9 baby skunks" } }, "dogs": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 5.173780918121338, "1": 4.413113594055176, "2": 18.805801391601562, "3": 2.291489601135254, "4": 0.4114401042461395, "5": 0.5587718486785889, "6": -2.013409376144409, "7": -2.6559765338897705, "8": -3.0427844524383545, "9": -0.9144049286842346 }, "allowed_token_mass": 1.0, "input_tokens": 608 }, "confidence": 0.9999878733557181, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 1.2014008220955e-06, "1": 5.614800157016732e-07, "2": 0.9999981432326194, "3": 6.728599989734901e-08, "4": 1.0266669659507874e-08, "5": 1.1896383391613673e-08, "6": 9.085123589339909e-10, "7": 4.77823459971288e-10, "8": 3.2454799027580044e-10, "9": 2.7266061297352917e-09 }, "type": "score", "score": 1.9999971862835273, "legend": { "0": "0 dogs", "1": "1 dogs", "2": "2 dogs", "3": "3 dogs", "4": "4 dogs", "5": "5 dogs", "6": "6 dogs", "7": "7 dogs", "8": "8 dogs", "9": "9 dogs" } }, "call_vet": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": 3.5363476276397705, "false": -2.9451053142547607 }, "allowed_token_mass": 0.9966760277748108, "input_tokens": 609 }, "type": "noul", "noul": 0.998470758401882 }, "action": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "Call the vet": -24.56247901916504, "Do not call the vet": -42.6781005859375 }, "allowed_token_mass": 2.1510519839546828e-11, "input_tokens": 613 }, "confidence": 0.9999996258476523, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "Call the vet": 0.9999999864329474, "Do not call the vet": 1.3567052682170536e-08 }, "type": "choice", "choice": "Call the vet", "selected_probability": 0.9999999864329474 } }, "inference_seconds": 0.28978804100188427, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 604, "primitive_fields": 4, "question_suffix_tokens": [ 5, 4, 5, 4 ], "candidate_token_lengths": [ [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1 ], [ 4, 6 ] ], "preprocess_seconds": 0.04708162500173785, "prefill_seconds": 0.1582839589973446, "branch_seconds": 0.08421125001041219, "branch_batch_sizes": [ 3 ], "candidate_batch_sizes": [ 2 ], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 0.28978804100188427 }, "comparison": { "seconds": { "parallel": 0.28978804100188427, "normal": 0.7000212080019992 }, "normal_over_parallel": null, "parallel_values": { "skunks": 0, "dogs": 2, "call_vet": true, "action": "Call the vet" }, "normal": { "answers": null, "raw_text": "{\"skunks\": \"0\", \"dogs\": \"2\", \"call_vet\": false, \"action\": \"Do not call the vet\"}", "valid": false, "error": "Generated value for skunks does not match its answer contract", "finish_reason": "stop", "input_tokens": 604, "output_tokens": 31, "max_output_tokens": 128, "preprocess_seconds": 0.03601499999058433, "generation_seconds": 0.6639642080117483, "inference_seconds": 0.7000212080019992, "total_seconds": 0.7000212080019992 }, "agreement": { "skunks": null, "dogs": null, "call_vet": null, "action": null }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "video_separate_speech", "expected": { "visible_skunks": 5, "spoken_dogs": 2, "call_vet": true }, "result": { "answers": { "visible_skunks": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 21.462350845336914, "1": 18.82436752319336, "2": 18.677799224853516, "3": 18.989131927490234, "4": 19.2896671295166, "5": 18.753393173217773, "6": 18.920808792114258, "7": 16.883216857910156, "8": 17.32362937927246, "9": 16.478431701660156 }, "allowed_token_mass": 0.999992311000824, "input_tokens": 869 }, "confidence": 0.4344185836325999, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 0.6623165780607772, "1": 0.047359163795677915, "2": 0.04090253476156381, "3": 0.055841914223619, "4": 0.0754190533772143, "5": 0.04411438783931597, "6": 0.05215403769027509, "7": 0.0067978723179101695, "8": 0.010559460523229133, "9": 0.004534997410417384 }, "type": "score", "score": 1.304738121941711, "legend": { "0": "0 baby skunks", "1": "1 baby skunks", "2": "2 baby skunks", "3": "3 baby skunks", "4": "4 baby skunks", "5": "5 baby skunks", "6": "6 baby skunks", "7": "7 baby skunks", "8": "8 baby skunks", "9": "9 baby skunks" } }, "spoken_dogs": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 17.806127548217773, "1": 19.231237411499023, "2": 23.706363677978516, "3": 16.521085739135742, "4": 15.655125617980957, "5": 13.830660820007324, "6": 14.085387229919434, "7": 14.52027702331543, "8": 15.201702117919922, "9": 14.943788528442383 }, "allowed_token_mass": 0.9999732375144958, "input_tokens": 868 }, "confidence": 0.958860059083027, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 0.0026962428593753284, "1": 0.011211826219185275, "2": 0.9844620994930602, "3": 0.0007458859751998891, "4": 0.0003137550604088172, "5": 5.0609930963419266e-05, "6": 6.529230777125973e-05, "7": 0.000100863087149435, "8": 0.00019937532807359187, "9": 0.00015404973881271634 }, "type": "score", "score": 1.9879609987579345, "legend": { "0": "0 dogs", "1": "1 dogs", "2": "2 dogs", "3": "3 dogs", "4": "4 dogs", "5": "5 dogs", "6": "6 dogs", "7": "7 dogs", "8": "8 dogs", "9": "9 dogs" } }, "call_vet": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "true": 7.686960697174072, "false": 6.179104804992676 }, "allowed_token_mass": 0.9969248175621033, "input_tokens": 867 }, "type": "noul", "noul": 0.8187432327795501 } }, "inference_seconds": 0.8796733749913983, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 862, "primitive_fields": 3, "question_suffix_tokens": [ 7, 6, 5 ], "candidate_token_lengths": [ [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1 ] ], "preprocess_seconds": 0.3215727080241777, "prefill_seconds": 0.5237813749990892, "branch_seconds": 0.03414949998841621, "branch_batch_sizes": [ 3 ], "candidate_batch_sizes": [], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 0.8796733749913983 }, "comparison": { "seconds": { "parallel": 0.8796733749913983, "normal": 1.1185389169841073 }, "normal_over_parallel": 1.2715389015781509, "parallel_values": { "visible_skunks": 0, "spoken_dogs": 2, "call_vet": true }, "normal": { "answers": { "visible_skunks": 0, "spoken_dogs": 2, "call_vet": true }, "raw_text": "{\"visible_skunks\": 0, \"spoken_dogs\": 2, \"call_vet\": true}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 862, "output_tokens": 25, "max_output_tokens": 128, "preprocess_seconds": 0.17747870800667442, "generation_seconds": 0.9410260419826955, "inference_seconds": 1.1185389169841073, "total_seconds": 1.1185389169841073 }, "agreement": { "visible_skunks": true, "spoken_dogs": true, "call_vet": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } }, { "name": "skunks_full_soundtrack", "expected": null, "result": { "answers": { "Baby skunks": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 15.440519332885742, "1": 18.91536521911621, "2": 17.52647590637207, "3": 17.263118743896484, "4": 17.27484130859375, "5": 25.10904312133789, "6": 19.844194412231445, "7": 18.685335159301758, "8": 18.2138671875, "9": 19.023639678955078 }, "allowed_token_mass": 0.9999809265136719, "input_tokens": 2442 }, "confidence": 0.9591435032147506, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 6.240175840612261e-05, "1": 0.002015130710062158, "2": 0.0005024770805513657, "3": 0.000386137245892033, "4": 0.00039069039991983777, "5": 0.9866959764138694, "6": 0.005101391037174207, "7": 0.0016010409348640024, "8": 0.0009991863613793212, "9": 0.002245568057881544 }, "type": "score", "score": 5.00924037645693, "legend": { "0": "0 baby skunks", "1": "1 baby skunks", "2": "2 baby skunks", "3": "3 baby skunks", "4": "4 baby skunks", "5": "5 baby skunks", "6": "6 baby skunks", "7": "7 baby skunks", "8": "8 baby skunks", "9": "9 baby skunks" } }, "Dog": { "calibration_status": "unvalidated", "temperature": 1.0, "probability_source": "restricted_json_value_likelihoods", "diagnostics": { "logits": { "0": 16.827749252319336, "1": 24.871902465820312, "2": 18.862998962402344, "3": 17.827789306640625, "4": 18.372800827026367, "5": 17.98102569580078, "6": 16.684083938598633, "7": 16.508058547973633, "8": 17.43865203857422, "9": 19.051477432250977 }, "allowed_token_mass": 0.9999866485595703, "input_tokens": 2440 }, "confidence": 0.9671310741513678, "confidence_definition": "one_minus_normalized_entropy", "probabilities": { "0": 0.0003177193855978654, "1": 0.9898629433627016, "2": 0.0024318760318991664, "3": 0.000863685426060581, "4": 0.0014895362854541818, "5": 0.0010067121486512969, "6": 0.00027520141403579993, "7": 0.0002307829950513383, "8": 0.0005852688675128792, "9": 0.0029362740830353746 }, "type": "score", "score": 1.0426847647267505, "legend": { "0": "0 dogs", "1": "1 dogs", "2": "2 dogs", "3": "3 dogs", "4": "4 dogs", "5": "5 dogs", "6": "6 dogs", "7": "7 dogs", "8": "8 dogs", "9": "9 dogs" } } }, "inference_seconds": 3.170949292019941, "execution": { "execution": "shared_json_prefix_gpu_batched_fields", "prefix_prefills": 1, "prefix_tokens": 2436, "primitive_fields": 2, "question_suffix_tokens": [ 6, 4 ], "candidate_token_lengths": [ [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ], [ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1 ] ], "preprocess_seconds": 0.6641687910014298, "prefill_seconds": 2.471621042001061, "branch_seconds": 0.03497962499386631, "branch_batch_sizes": [ 2 ], "candidate_batch_sizes": [], "compute_dtype": "float32", "kv_storage": "replicated_per_batch_row", "conditioning": "shared complete question schema; no previous field answers" }, "valid": true, "total_seconds": 3.170949292019941 }, "comparison": { "seconds": { "parallel": 3.170949292019941, "normal": 3.328362667001784 }, "normal_over_parallel": 1.0496423501246115, "parallel_values": { "Baby skunks": 5, "Dog": 1 }, "normal": { "answers": { "Baby skunks": 5, "Dog": 1 }, "raw_text": "{\"Baby skunks\": 5, \"Dog\": 1}", "valid": true, "error": null, "finish_reason": "stop", "input_tokens": 2436, "output_tokens": 15, "max_output_tokens": 128, "preprocess_seconds": 0.653564042004291, "generation_seconds": 2.6747562079981435, "inference_seconds": 3.328362667001784, "total_seconds": 3.328362667001784 }, "agreement": { "Baby skunks": true, "Dog": true }, "methodology": { "model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers", "timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.", "cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.", "answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement." } } } ] }