gemma-e2b-rlcd / reports /native-json-web-validation.json
larkooo's picture
Publish Gemma E2B RLCD with multimodal checkpoint and parallel scoring
53e24ca verified
Raw History Blame Contribute Delete
19.8 kB
{
"scope": "Local web runtime integration checks, not a general accuracy or speed benchmark. Full-video expected values are prior baseline outputs, not independently annotated ground truth.",
"cases": [
{
"name": "skunks_full_soundtrack",
"spec": {
"text": "",
"questions": {
"Baby skunks": {
"type": "score",
"instructions": "Count the baby skunks visible at the same time in this video. Do not count the same animal again across frames. Choose the closest count.",
"criteria": [
"0 baby skunks",
"1 baby skunks",
"2 baby skunks",
"3 baby skunks",
"4 baby skunks",
"5 baby skunks",
"6 baby skunks",
"7 baby skunks",
"8 baby skunks",
"9 baby skunks"
]
},
"Dog": {
"type": "score",
"instructions": "Count the dogs visible at the same time in this video. Do not count the same animal again across frames. Choose the closest count.",
"criteria": [
"0 dogs",
"1 dogs",
"2 dogs",
"3 dogs",
"4 dogs",
"5 dogs",
"6 dogs",
"7 dogs",
"8 dogs",
"9 dogs"
]
}
},
"media": [
{
"kind": "video",
"name": "skunks-full.mp4"
}
]
},
"media_sha256": "94860ab16790f8a4e6ae9c86c97e35322e1e4f58e2695c593dca0d5c5b170148",
"expected": {
"Baby skunks": 5,
"Dog": 1
},
"result": {
"answers": {
"Baby skunks": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_json_value_likelihoods",
"diagnostics": {
"logits": {
"0": 15.440519332885742,
"1": 18.91536521911621,
"2": 17.52647590637207,
"3": 17.263118743896484,
"4": 17.27484130859375,
"5": 25.10904312133789,
"6": 19.844194412231445,
"7": 18.685335159301758,
"8": 18.2138671875,
"9": 19.023639678955078
},
"allowed_token_mass": 0.9999809265136719,
"input_tokens": 2442
},
"confidence": 0.9591435032147506,
"confidence_definition": "one_minus_normalized_entropy",
"probabilities": {
"0": 6.240175840612261e-05,
"1": 0.002015130710062158,
"2": 0.0005024770805513657,
"3": 0.000386137245892033,
"4": 0.00039069039991983777,
"5": 0.9866959764138694,
"6": 0.005101391037174207,
"7": 0.0016010409348640024,
"8": 0.0009991863613793212,
"9": 0.002245568057881544
},
"type": "score",
"score": 5.00924037645693,
"legend": {
"0": "0 baby skunks",
"1": "1 baby skunks",
"2": "2 baby skunks",
"3": "3 baby skunks",
"4": "4 baby skunks",
"5": "5 baby skunks",
"6": "6 baby skunks",
"7": "7 baby skunks",
"8": "8 baby skunks",
"9": "9 baby skunks"
}
},
"Dog": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_json_value_likelihoods",
"diagnostics": {
"logits": {
"0": 16.827749252319336,
"1": 24.871902465820312,
"2": 18.862998962402344,
"3": 17.827789306640625,
"4": 18.372800827026367,
"5": 17.98102569580078,
"6": 16.684083938598633,
"7": 16.508058547973633,
"8": 17.43865203857422,
"9": 19.051477432250977
},
"allowed_token_mass": 0.9999866485595703,
"input_tokens": 2440
},
"confidence": 0.9671310741513678,
"confidence_definition": "one_minus_normalized_entropy",
"probabilities": {
"0": 0.0003177193855978654,
"1": 0.9898629433627016,
"2": 0.0024318760318991664,
"3": 0.000863685426060581,
"4": 0.0014895362854541818,
"5": 0.0010067121486512969,
"6": 0.00027520141403579993,
"7": 0.0002307829950513383,
"8": 0.0005852688675128792,
"9": 0.0029362740830353746
},
"type": "score",
"score": 1.0426847647267505,
"legend": {
"0": "0 dogs",
"1": "1 dogs",
"2": "2 dogs",
"3": "3 dogs",
"4": "4 dogs",
"5": "5 dogs",
"6": "6 dogs",
"7": "7 dogs",
"8": "8 dogs",
"9": "9 dogs"
}
}
},
"inference_seconds": 3.0780157500121277,
"execution": {
"execution": "shared_json_prefix_gpu_batched_fields",
"prefix_prefills": 1,
"prefix_tokens": 2436,
"primitive_fields": 2,
"question_suffix_tokens": [
6,
4
],
"candidate_token_lengths": [
[
1,
1,
1,
1,
1,
1,
1,
1,
1,
1
],
[
1,
1,
1,
1,
1,
1,
1,
1,
1,
1
]
],
"preprocess_seconds": 0.6728912920225412,
"prefill_seconds": 2.3704014160030056,
"branch_seconds": 0.0345094169897493,
"branch_batch_sizes": [
2
],
"candidate_batch_sizes": [],
"compute_dtype": "float32",
"kv_storage": "replicated_per_batch_row",
"conditioning": "shared complete question schema; no previous field answers"
},
"valid": true,
"total_seconds": 6.285174749995349,
"media": [
{
"name": "skunks-full.mp4",
"kind": "video",
"bytes": 2935259,
"duration_seconds": 21.161042,
"has_soundtrack": true
}
],
"media_prepare_seconds": 0.05148224998265505,
"decision_seconds": 3.0780157500121277,
"model": "Gemma 4 E2B \u00b7 frozen scorer",
"load_seconds": 3.548284250020515,
"comparison": {
"seconds": {
"parallel": 3.1294979999947827,
"normal": 3.179176165984245
},
"normal_over_parallel": 1.0158741644792695,
"parallel_values": {
"Baby skunks": 5,
"Dog": 1
},
"normal": {
"answers": {
"Baby skunks": 5,
"Dog": 1
},
"raw_text": "{\"Baby skunks\": 5, \"Dog\": 1}",
"valid": true,
"error": null,
"finish_reason": "stop",
"input_tokens": 2436,
"output_tokens": 15,
"max_output_tokens": 128,
"preprocess_seconds": 0.5861692079924978,
"generation_seconds": 2.5414814579999074,
"inference_seconds": 3.12769391600159,
"total_seconds": 3.179176165984245
},
"agreement": {
"Baby skunks": true,
"Dog": true
},
"methodology": {
"model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers",
"timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.",
"cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.",
"answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement."
}
},
"request_seconds": 6.292751375003718
}
},
{
"name": "speech_skunks",
"spec": {
"text": "",
"media": [
{
"kind": "audio",
"name": "skunks.wav"
}
],
"questions": {
"skunks": {
"type": "score",
"instructions": "How many baby skunks does the speaker say there are?",
"criteria": [
"0 baby skunks",
"1 baby skunks",
"2 baby skunks",
"3 baby skunks",
"4 baby skunks",
"5 baby skunks",
"6 baby skunks",
"7 baby skunks",
"8 baby skunks",
"9 baby skunks"
]
},
"dogs": {
"type": "score",
"instructions": "How many dogs does the speaker say there are?",
"criteria": [
"0 dogs",
"1 dogs",
"2 dogs",
"3 dogs",
"4 dogs",
"5 dogs",
"6 dogs",
"7 dogs",
"8 dogs",
"9 dogs"
]
},
"call_vet": {
"type": "noul",
"instructions": "Does the speaker request that we call the vet?"
},
"action": {
"type": "choice",
"instructions": "What does the speaker ask us to do?",
"criteria": {
"Call the vet": "Call the vet",
"Do not call the vet": "Do not call the vet"
}
}
}
},
"media_sha256": "2b4e20e573eab19383d212d0bc718f97c0ad2da1c9c9120c506a5f17753abddc",
"expected": {
"skunks": 5,
"dogs": 1,
"call_vet": false,
"action": "Do not call the vet"
},
"result": {
"answers": {
"skunks": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_json_value_likelihoods",
"diagnostics": {
"logits": {
"0": 2.4332430362701416,
"1": 0.33744150400161743,
"2": 0.49974024295806885,
"3": 1.0511709451675415,
"4": 2.741050958633423,
"5": 20.806188583374023,
"6": 2.202913999557495,
"7": -1.8478120565414429,
"8": -2.2469234466552734,
"9": 0.48146724700927734
},
"allowed_token_mass": 1.0,
"input_tokens": 620
},
"confidence": 0.9999996565698799,
"confidence_definition": "one_minus_normalized_entropy",
"probabilities": {
"0": 1.0488928261791463e-08,
"1": 1.2898406934682289e-09,
"2": 1.5171255458984183e-09,
"3": 2.6333272442131735e-09,
"4": 1.4269553933267914e-08,
"5": 0.9999999597381529,
"6": 8.331064281204806e-09,
"7": 1.4504157649251244e-10,
"8": 9.731070908143261e-11,
"9": 1.4896548671182722e-09
},
"type": "score",
"score": 4.999999933180111,
"legend": {
"0": "0 baby skunks",
"1": "1 baby skunks",
"2": "2 baby skunks",
"3": "3 baby skunks",
"4": "4 baby skunks",
"5": "5 baby skunks",
"6": "6 baby skunks",
"7": "7 baby skunks",
"8": "8 baby skunks",
"9": "9 baby skunks"
}
},
"dogs": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_json_value_likelihoods",
"diagnostics": {
"logits": {
"0": 12.110952377319336,
"1": 25.3037166595459,
"2": 9.88134765625,
"3": 6.69423246383667,
"4": 8.310088157653809,
"5": 7.257328510284424,
"6": 6.1623053550720215,
"7": 4.915668487548828,
"8": 5.334745407104492,
"9": 7.838923931121826
},
"allowed_token_mass": 1.0,
"input_tokens": 619
},
"confidence": 0.9999862804353773,
"confidence_definition": "one_minus_normalized_entropy",
"probabilities": {
"0": 1.86403615456672e-06,
"1": 0.9999978365667048,
"2": 2.005161255538126e-07,
"3": 8.279474314556474e-09,
"4": 4.16639052158653e-08,
"5": 1.4539593643851128e-08,
"6": 4.863957315537724e-09,
"7": 1.3982416759546908e-09,
"8": 2.126106602756648e-09,
"9": 2.6009736406271668e-08
},
"type": "score",
"score": 0.9999987918588841,
"legend": {
"0": "0 dogs",
"1": "1 dogs",
"2": "2 dogs",
"3": "3 dogs",
"4": "4 dogs",
"5": "5 dogs",
"6": "6 dogs",
"7": "7 dogs",
"8": "8 dogs",
"9": "9 dogs"
}
},
"call_vet": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_json_value_likelihoods",
"diagnostics": {
"logits": {
"true": -0.8286030292510986,
"false": 6.238976955413818
},
"allowed_token_mass": 0.999083399772644,
"input_tokens": 620
},
"type": "noul",
"noul": 0.0008515673929433828
},
"action": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_json_value_likelihoods",
"diagnostics": {
"logits": {
"Call the vet": -40.304771423339844,
"Do not call the vet": -23.257143020629883
},
"allowed_token_mass": 7.93507687306582e-11,
"input_tokens": 624
},
"confidence": 0.9999989722115856,
"confidence_definition": "one_minus_normalized_entropy",
"probabilities": {
"Call the vet": 3.947380923962134e-08,
"Do not call the vet": 0.9999999605261907
},
"type": "choice",
"choice": "Do not call the vet",
"selected_probability": 0.9999999605261907
}
},
"inference_seconds": 0.27987883301102556,
"execution": {
"execution": "shared_json_prefix_gpu_batched_fields",
"prefix_prefills": 1,
"prefix_tokens": 615,
"primitive_fields": 4,
"question_suffix_tokens": [
5,
4,
5,
4
],
"candidate_token_lengths": [
[
1,
1,
1,
1,
1,
1,
1,
1,
1,
1
],
[
1,
1,
1,
1,
1,
1,
1,
1,
1,
1
],
[
1,
1
],
[
4,
6
]
],
"preprocess_seconds": 0.043527457979507744,
"prefill_seconds": 0.153558875026647,
"branch_seconds": 0.08244479197310284,
"branch_batch_sizes": [
3
],
"candidate_batch_sizes": [
2
],
"compute_dtype": "float32",
"kv_storage": "replicated_per_batch_row",
"conditioning": "shared complete question schema; no previous field answers"
},
"valid": true,
"total_seconds": 1.0800352079968434,
"media": [
{
"name": "skunks.wav",
"kind": "audio",
"bytes": 148568,
"duration_seconds": 4.51475,
"has_soundtrack": false
}
],
"media_prepare_seconds": 0.10681216599186882,
"decision_seconds": 0.27987883301102556,
"model": "Gemma 4 E2B \u00b7 frozen scorer",
"load_seconds": 3.548284250020515,
"comparison": {
"seconds": {
"parallel": 0.3866909990028944,
"normal": 0.7924858739716001
},
"normal_over_parallel": 2.0494034668897694,
"parallel_values": {
"skunks": 5,
"dogs": 1,
"call_vet": false,
"action": "Do not call the vet"
},
"normal": {
"answers": {
"skunks": 5,
"dogs": 1,
"call_vet": false,
"action": "Do not call the vet"
},
"raw_text": "{\"skunks\": 5, \"dogs\": 1, \"call_vet\": false, \"action\": \"Do not call the vet\"}",
"valid": true,
"error": null,
"finish_reason": "stop",
"input_tokens": 615,
"output_tokens": 31,
"max_output_tokens": 128,
"preprocess_seconds": 0.03355474999989383,
"generation_seconds": 0.6520849579828791,
"inference_seconds": 0.6856737079797313,
"total_seconds": 0.7924858739716001
},
"agreement": {
"skunks": true,
"dogs": true,
"call_vet": true,
"action": true
},
"methodology": {
"model": "Same resident Gemma 4 E2B 4-bit weights; float32 compute; all 35 layers",
"timing": "One run per path, parallel scorer first, normal generation second. Includes input preparation and inference; shared upload decoding added equally to each path. Excludes model loading, upload transfer, and allocator reset. No warm-up runs; first-use effects and run order can affect this observation.",
"cache": "Fresh input KV and media features for every run; ordinary output-token KV caching remains enabled for normal generation.",
"answers": "Normal Gemma generates one JSON object for all fields, greedily, without thinking. Compare choice names, most likely grade levels, and booleans at a 50% threshold. The scorer also returns probability distributions and expected grades. Agreement is not an accuracy measurement."
}
},
"request_seconds": 1.0820253329875413
}
}
]
}