Publish nested-reference MP-OPD probe 01c7bf4d
Browse files- MP-OPD-MULTIREF.md +49 -0
- simct-b200-portable.bundle +2 -2
- source-manifest.json +9 -3
MP-OPD-MULTIREF.md
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Nested 1/4/8-reference mechanics probe
|
| 2 |
+
|
| 3 |
+
Prepare4 groups of13 unique prompts: one rollout,8 select,4 eval. Select pools
|
| 4 |
+
are nested prefixes1/4/8, predetermined by a preparation seed, never chosen by
|
| 5 |
+
eval. Mean outer NLL equally weights references; each reference is token-mean.
|
| 6 |
+
One rollout and teacher credit computation per group, reused across counts.
|
| 7 |
+
Frozen adapter and model are unchanged across candidates. Weighting networks
|
| 8 |
+
start identically and train independently per count on earlier select groups.
|
| 9 |
+
Eval never trains or selects; all count outcomes are reported. Four groups
|
| 10 |
+
remain four independent units, not12. The ordinary summary describes the
|
| 11 |
+
largest count; paired-select-counts.json and results.jsonl include all counts.
|
| 12 |
+
Checkpoint weighting.pt is the largest-count network; weighting-select-N.pt
|
| 13 |
+
files retain each count. This is not a resume-complete training checkpoint.
|
| 14 |
+
|
| 15 |
+
## Company run, only after a new HF bundle pull
|
| 16 |
+
|
| 17 |
+
From the new isolated source checkout returned by the existing HF pull flow:
|
| 18 |
+
|
| 19 |
+
```bash
|
| 20 |
+
bash "$ABC_WORK/source/experiments/runai/transfer/04-prepare-multiref.sh"
|
| 21 |
+
# Explicit GPU boundary; only after the card is free and not reserved by eval:
|
| 22 |
+
bash "$ABC_WORK/source/experiments/runai/transfer/05-run-multiref.sh" 1
|
| 23 |
+
```
|
| 24 |
+
|
| 25 |
+
Prepare reuses the pinned selected.parquet or downloads its pinned HF revision
|
| 26 |
+
if missing, verifies SHA256, excludes all conflicting-reference prompt groups,
|
| 27 |
+
and writes a new exclusive file. This default source potentially appeared in
|
| 28 |
+
SFT: results are mechanics only, not unseen-data generalization evidence.
|
| 29 |
+
All52 selected prompts are disjoint across roles/groups, but no claim of
|
| 30 |
+
semantic deduplication or exclusion from external model pretraining is made.
|
| 31 |
+
For a scientific test, prepare a separate authorized train/dev reference source
|
| 32 |
+
and use --exclude-prompts for SFT/train/prior probe sets. Never use benchmark
|
| 33 |
+
test references. Keep the evaluation reference set fixed across select counts.
|
| 34 |
+
|
| 35 |
+
Runner uses exact company SFT Gemma and Qwen7B paths, FP32, maxnew64,
|
| 36 |
+
virtualLR.1, maxref1024,15-minute timeout. Long references fail rather than
|
| 37 |
+
truncate. It refuses the eval queue's GPU lock or >=1GiB occupied VRAM, so do
|
| 38 |
+
not stop eval to make room. A timeout preserves partial logs and returns124;
|
| 39 |
+
partial groups are not efficacy evidence. No automatic W&B upload.
|
| 40 |
+
|
| 41 |
+
## Validation
|
| 42 |
+
|
| 43 |
+
CPU fake-HF end-to-end tests cover legacy and1/4/8 modes, unchanged real
|
| 44 |
+
parameters, identical atomic eval NLL across counts, fixed eval IDs, nested
|
| 45 |
+
select counts, independent group accounting, overlap rejection, and analytical
|
| 46 |
+
gradients of the mean-reference objective. The multi-reference path completed on Modal A100-80GB (lhtu05), app
|
| 47 |
+
ap-uR42YdE7fVsJg3tbfZYE5h, source ee28346: four synthetic groups, base
|
| 48 |
+
Gemma-2-2b-it / Phi-4-mini, FP32, virtual LR 0.1. Company SFT/Qwen remains
|
| 49 |
+
unexecuted. See MP_OPD_OVERNIGHT_20260910.md for results and limitations.
|
simct-b200-portable.bundle
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4872eae417f41ad55a8a46e03aba975d9362f9023b977825cce3cf86030cf6e7
|
| 3 |
+
size 4184862
|
source-manifest.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
"source_branch": "vdt/ops/b200-portable",
|
| 3 |
-
"source_commit": "
|
| 4 |
"bundle_file": "simct-b200-portable.bundle",
|
| 5 |
-
"bundle_sha256": "
|
| 6 |
-
"bundle_bytes":
|
| 7 |
"history": "complete",
|
| 8 |
"format": "git-bundle",
|
| 9 |
"reason": "HF Git receive hook rejects historical PNG/JPG blobs; bundle preserves original Git object IDs",
|
|
@@ -21,5 +21,11 @@
|
|
| 21 |
"eval_queue": {
|
| 22 |
"pull-eval-queue.sh": "d20d18a6b22bae31313f13b26252f7032a290e4b229af842ddfe3a0d77ad8aa8",
|
| 23 |
"EVAL-QUEUE-RUNBOOK.md": "e249ed6750fb4264496fd6ec7bf2a5dcc67ccfc3be4e7f85be05cb00d8446f2b"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
}
|
| 25 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"source_branch": "vdt/ops/b200-portable",
|
| 3 |
+
"source_commit": "01c7bf4d591271643f9ab9397cc0b4916059ee31",
|
| 4 |
"bundle_file": "simct-b200-portable.bundle",
|
| 5 |
+
"bundle_sha256": "4872eae417f41ad55a8a46e03aba975d9362f9023b977825cce3cf86030cf6e7",
|
| 6 |
+
"bundle_bytes": 4184862,
|
| 7 |
"history": "complete",
|
| 8 |
"format": "git-bundle",
|
| 9 |
"reason": "HF Git receive hook rejects historical PNG/JPG blobs; bundle preserves original Git object IDs",
|
|
|
|
| 21 |
"eval_queue": {
|
| 22 |
"pull-eval-queue.sh": "d20d18a6b22bae31313f13b26252f7032a290e4b229af842ddfe3a0d77ad8aa8",
|
| 23 |
"EVAL-QUEUE-RUNBOOK.md": "e249ed6750fb4264496fd6ec7bf2a5dcc67ccfc3be4e7f85be05cb00d8446f2b"
|
| 24 |
+
},
|
| 25 |
+
"multiref_probe": {
|
| 26 |
+
"runbook": "MP-OPD-MULTIREF.md",
|
| 27 |
+
"sha256": "ae03b4e845d4dfe2dbb060a7015529d99cc837248926373da71d65441aef40cf",
|
| 28 |
+
"gpu_canary_source": "ee28346",
|
| 29 |
+
"company_gpu_run": false
|
| 30 |
}
|
| 31 |
}
|