codegeist commited on
Commit
312a68f
·
verified ·
1 Parent(s): 709dcff

Verify strict GPU-only model state

Browse files

Record the final A10G proof that all parameters and buffers are on CUDA, all floating parameters are BF16, and the raw response matches exactly.

Files changed (4) hide show
  1. README.md +8 -5
  2. SHA256SUMS +3 -3
  3. gpu-test-result.json +6 -4
  4. publication.json +28 -2
README.md CHANGED
@@ -172,22 +172,25 @@ inference repeatability, deterministic PyTorch algorithms, coding benchmarks,
172
  safety evaluation, and generalization were not tested.
173
 
174
  The successful public-artifact verification ran as Hugging Face Job
175
- [`6a760e12da2af92a634eedc6`](https://huggingface.co/jobs/codegeist/6a760e12da2af92a634eedc6)
176
  on NVIDIA A10G. The Job received no secrets and loaded the public base and
177
  adapter commits with implicit token use disabled. It verified:
178
 
179
  - Adapter weight SHA-256
180
  `19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8`.
181
- - CUDA BF16 with every model parameter on the GPU and no CPU fallback.
 
182
  - Peak allocated CUDA memory of 3,511,419,904 bytes.
183
- - A 21.724-second measured load-and-generation phase.
184
  - Exact raw and whitespace-normalized response
185
  `Codegeist is a coding agent.`.
186
 
187
  `gpu-test-result.json` contains the sanitized result and source hashes. The Job
188
- ran for 75 reported seconds. An earlier 92-second publication test failed before
189
  adapter injection because the Unsloth training lock includes TorchAO 0.13, which
190
- direct PEFT 0.20 inference rejects. The successful test used a separate locked
 
 
191
  inference environment without Unsloth or TorchAO; the adapter is not
192
  TorchAO-quantized. CPU inference remains outside the supported contract.
193
 
 
172
  safety evaluation, and generalization were not tested.
173
 
174
  The successful public-artifact verification ran as Hugging Face Job
175
+ [`6a7610a53e1f34a7e32bd8a8`](https://huggingface.co/jobs/codegeist/6a7610a53e1f34a7e32bd8a8)
176
  on NVIDIA A10G. The Job received no secrets and loaded the public base and
177
  adapter commits with implicit token use disabled. It verified:
178
 
179
  - Adapter weight SHA-256
180
  `19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8`.
181
+ - CUDA BF16 with every floating parameter in BF16 and every parameter and buffer
182
+ on the GPU, with no CPU fallback.
183
  - Peak allocated CUDA memory of 3,511,419,904 bytes.
184
+ - A 20.069-second measured load-and-generation phase.
185
  - Exact raw and whitespace-normalized response
186
  `Codegeist is a coding agent.`.
187
 
188
  `gpu-test-result.json` contains the sanitized result and source hashes. The Job
189
+ ran for 76 reported seconds. An earlier 92-second publication test failed before
190
  adapter injection because the Unsloth training lock includes TorchAO 0.13, which
191
+ direct PEFT 0.20 inference rejects. A preliminary 75-second pass then verified
192
+ all parameters on CUDA; the final Job expanded the gate to every buffer and
193
+ every floating-parameter dtype. The successful tests used a separate locked
194
  inference environment without Unsloth or TorchAO; the adapter is not
195
  TorchAO-quantized. CPU inference remains outside the supported contract.
196
 
SHA256SUMS CHANGED
@@ -1,8 +1,8 @@
1
  9a66ed1f77d750a879b0e7b610bb15bb7c109fc1158448c0d7d543e7dbef421f LICENSE
2
- c66441fa829c3fd05222fa22d2f6307903bf35ba7652d2df42617367e3a58af0 README.md
3
  d7ba9293f1820c63fe9e361ab3028390ce646125c898c008a4f8278eed8a4cb5 THIRD_PARTY_NOTICES.md
4
  42ef1e8075588b3732d0c00dd4c2a08a5e3498429d0e83192210e8006bdf15fb adapter_config.json
5
  19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8 adapter_model.safetensors
6
  502de0994cc784811c28aeb2bb27b478887256489764417543b7493dfd44f7c6 evidence.json
7
- c5b3e8567fc77050e6074ca944cb5ffca1603b7072d27dca69df9b9c67727939 gpu-test-result.json
8
- 25aaf0ce11772c9466f907c7b1495e981802ef352eac42beedaae42c85ad0c67 publication.json
 
1
  9a66ed1f77d750a879b0e7b610bb15bb7c109fc1158448c0d7d543e7dbef421f LICENSE
2
+ 21f2832009be8bb1e8095405467d18fb73ddb1d94eb49230bbfd39108609e5e5 README.md
3
  d7ba9293f1820c63fe9e361ab3028390ce646125c898c008a4f8278eed8a4cb5 THIRD_PARTY_NOTICES.md
4
  42ef1e8075588b3732d0c00dd4c2a08a5e3498429d0e83192210e8006bdf15fb adapter_config.json
5
  19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8 adapter_model.safetensors
6
  502de0994cc784811c28aeb2bb27b478887256489764417543b7493dfd44f7c6 evidence.json
7
+ 339a15a527229ab82bebce069cb96987a6e2ebb977261f03553759a8f979e57a gpu-test-result.json
8
+ 626d130fd93c5afcca83ef5e3f25cc1012eb1bf7705cc67972c8e08148f3c358 publication.json
gpu-test-result.json CHANGED
@@ -2,17 +2,19 @@
2
  "adapter_model": "codegeist/qwen3-1.7b-codegeist-identity-smoke",
3
  "adapter_revision": "04d51edac56c6f1e068c644bfa8d014cadcecf9f",
4
  "adapter_weight_sha256": "19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8",
 
 
5
  "all_parameters_on_cuda": true,
6
  "base_model": "Qwen/Qwen3-1.7B",
 
7
  "base_revision": "70d244cc86ccca08cf5af4e1e306ecf908b1ad5e",
8
  "device": "cuda",
9
- "dtype": "bfloat16",
10
- "duration_seconds": 21.724,
11
  "expected_response": "Codegeist is a coding agent.",
12
  "hardware": "NVIDIA A10G",
13
  "job": {
14
  "accelerator": "gpu",
15
- "id": "6a760e12da2af92a634eedc6"
16
  },
17
  "normalization": "strip leading and trailing whitespace",
18
  "normalized_match": true,
@@ -32,7 +34,7 @@
32
  "python": "3.12.12"
33
  },
34
  "source_sha256": {
35
- "infer.py": "b540a6f584e65fcf5c13811a5133b7477965e5bf6764645cc67620a2728cb50d",
36
  "inference/pyproject.toml": "b027bca31339345c4ba5ad886952e3b724f05d936df3fb220ef2d0af99783ea4",
37
  "inference/uv.lock": "ebeda66f1193fbdddd4a06c7e3ac3c7789d78c84c224259246e43214b7031bfa"
38
  }
 
2
  "adapter_model": "codegeist/qwen3-1.7b-codegeist-identity-smoke",
3
  "adapter_revision": "04d51edac56c6f1e068c644bfa8d014cadcecf9f",
4
  "adapter_weight_sha256": "19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8",
5
+ "all_buffers_on_cuda": true,
6
+ "all_floating_parameters_bfloat16": true,
7
  "all_parameters_on_cuda": true,
8
  "base_model": "Qwen/Qwen3-1.7B",
9
+ "base_model_dtype": "bfloat16",
10
  "base_revision": "70d244cc86ccca08cf5af4e1e306ecf908b1ad5e",
11
  "device": "cuda",
12
+ "duration_seconds": 20.069,
 
13
  "expected_response": "Codegeist is a coding agent.",
14
  "hardware": "NVIDIA A10G",
15
  "job": {
16
  "accelerator": "gpu",
17
+ "id": "6a7610a53e1f34a7e32bd8a8"
18
  },
19
  "normalization": "strip leading and trailing whitespace",
20
  "normalized_match": true,
 
34
  "python": "3.12.12"
35
  },
36
  "source_sha256": {
37
+ "infer.py": "f5a4c47cf9362ec9bfd3f119f8829f59e9691d426ab503b83423110a2e1aa553",
38
  "inference/pyproject.toml": "b027bca31339345c4ba5ad886952e3b724f05d936df3fb220ef2d0af99783ea4",
39
  "inference/uv.lock": "ebeda66f1193fbdddd4a06c7e3ac3c7789d78c84c224259246e43214b7031bfa"
40
  }
publication.json CHANGED
@@ -25,7 +25,7 @@
25
  "running_seconds": 92,
26
  "finding": "The Unsloth training lock installs TorchAO 0.13, which direct PEFT 0.20 adapter injection rejects."
27
  },
28
- "successful_job": {
29
  "id": "6a760e12da2af92a634eedc6",
30
  "terminal_status": "COMPLETED",
31
  "running_seconds": 75,
@@ -43,10 +43,36 @@
43
  "normalized_match": true,
44
  "result_sha256": "c5b3e8567fc77050e6074ca944cb5ffca1603b7072d27dca69df9b9c67727939"
45
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
46
  "inference_source_sha256": {
47
- "infer.py": "b540a6f584e65fcf5c13811a5133b7477965e5bf6764645cc67620a2728cb50d",
48
  "inference/pyproject.toml": "b027bca31339345c4ba5ad886952e3b724f05d936df3fb220ef2d0af99783ea4",
49
  "inference/uv.lock": "ebeda66f1193fbdddd4a06c7e3ac3c7789d78c84c224259246e43214b7031bfa"
 
 
 
 
 
 
50
  }
51
  },
52
  "adapter_weights_changed": false,
 
25
  "running_seconds": 92,
26
  "finding": "The Unsloth training lock installs TorchAO 0.13, which direct PEFT 0.20 adapter injection rejects."
27
  },
28
+ "preliminary_successful_job": {
29
  "id": "6a760e12da2af92a634eedc6",
30
  "terminal_status": "COMPLETED",
31
  "running_seconds": 75,
 
43
  "normalized_match": true,
44
  "result_sha256": "c5b3e8567fc77050e6074ca944cb5ffca1603b7072d27dca69df9b9c67727939"
45
  },
46
+ "successful_job": {
47
+ "id": "6a7610a53e1f34a7e32bd8a8",
48
+ "terminal_status": "COMPLETED",
49
+ "running_seconds": 76,
50
+ "secrets": [],
51
+ "hardware": "NVIDIA A10G",
52
+ "device": "cuda",
53
+ "base_model_dtype": "bfloat16",
54
+ "all_floating_parameters_bfloat16": true,
55
+ "all_parameters_on_cuda": true,
56
+ "all_buffers_on_cuda": true,
57
+ "peak_cuda_memory_bytes": 3511419904,
58
+ "measured_phase_seconds": 20.069,
59
+ "adapter_revision": "04d51edac56c6f1e068c644bfa8d014cadcecf9f",
60
+ "adapter_weight_sha256": "19d424106ef88ffeac4c26c22cebfb13ae1d5f309e1dcccf2da708727bec10a8",
61
+ "raw_response": "Codegeist is a coding agent.",
62
+ "normalized_response": "Codegeist is a coding agent.",
63
+ "normalized_match": true,
64
+ "result_sha256": "339a15a527229ab82bebce069cb96987a6e2ebb977261f03553759a8f979e57a"
65
+ },
66
  "inference_source_sha256": {
67
+ "infer.py": "f5a4c47cf9362ec9bfd3f119f8829f59e9691d426ab503b83423110a2e1aa553",
68
  "inference/pyproject.toml": "b027bca31339345c4ba5ad886952e3b724f05d936df3fb220ef2d0af99783ea4",
69
  "inference/uv.lock": "ebeda66f1193fbdddd4a06c7e3ac3c7789d78c84c224259246e43214b7031bfa"
70
+ },
71
+ "cost_estimate": {
72
+ "running_seconds": 243,
73
+ "per_second_estimate_usd": 0.0675,
74
+ "conservative_whole_minutes": 6,
75
+ "conservative_estimate_usd": 0.1002
76
  }
77
  },
78
  "adapter_weights_changed": false,