coding-agents
tool-selection
paintedwolf-ai commited on
Commit
49119f9
·
verified ·
1 Parent(s): 39012d0

Release open1-b5-b7g-e4

Browse files
Files changed (4) hide show
  1. PROVENANCE.json +52 -56
  2. README.md +41 -99
  3. SHA256SUMS +3 -6
  4. guide-load.safetensors +3 -0
PROVENANCE.json CHANGED
@@ -1,17 +1,55 @@
1
  {
2
- "schema": "pw-decide-heads-evidence/1",
3
- "version": "open1-b5-e4",
4
- "model_repo": "paintedwolfcode/bialy",
5
- "code_repo": "https://github.com/paintedwolf-ai/bialy",
6
- "backbone": {
7
- "model": "convaiinnovations/laya-multilingual",
8
- "revision": "e4e9ddf21a7b1903b7acffd8814ad4307bf63a67"
9
- },
10
  "heads": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11
  {
12
  "file": "turn-load.safetensors",
13
  "sha256": "550f30b94d5c6a8680d2cb2a607ebc9ee33e709c7679f72606d1e9fda17f2e90",
14
- "bytes": 59881652,
15
  "metadata": {
16
  "by_type": "{\"loss\": {\"tools\": 0.0732}, \"tools\": {\"precision\": 0.417, \"recall\": 0.691, \"positives\": 285, \"options\": 30083}, \"tools@coordinator\": {\"precision\": 0.422, \"recall\": 0.691, \"positives\": 285, \"options\": 27810}, \"tools@worker\": {\"precision\": 0.0, \"recall\": 0.0, \"positives\": 0, \"options\": 2273}}",
17
  "head_max_len": "512",
@@ -40,7 +78,6 @@
40
  {
41
  "file": "unit-rank.safetensors",
42
  "sha256": "c62674993eabba94660eccfafbb6e04d1549adc7335656c0ddefa7f068fa58f7",
43
- "bytes": 59881268,
44
  "metadata": {
45
  "val_loss": "0.58553760010303",
46
  "format": "pw-decide-head/1",
@@ -62,53 +99,12 @@
62
  "rank_levels": "skills-blended",
63
  "state_tokens": "{}"
64
  }
65
- },
66
- {
67
- "file": "code-rank.safetensors",
68
- "sha256": "abe235c347a223e957a0f8b7b979e3a584aebab76ee239e5f62117e839b9c40a",
69
- "bytes": 59880964,
70
- "metadata": {
71
- "label": "open1-code-rank",
72
- "model": "convaiinnovations/laya-multilingual",
73
- "val_loss": "0.734294557281827",
74
- "train_examples": "251360",
75
- "backbone": "jhu-clsp/mmBERT-base",
76
- "max_len": "512",
77
- "format": "pw-decide-head/1",
78
- "val_acc": "0.684346701164295",
79
- "val_mae": "0.7976714370407317"
80
- }
81
  }
82
  ],
83
- "release_manifest_sha256": "404f04b03fcf6b757330eda9c0841835248b1f8dc105fb3b61f01dfa9657c872",
84
- "dataset_version": "open1-b5-e4-bialy",
85
- "dataset_status": "A companion dataset bundle is prepared separately. This statement does not assert that either bundle has been published.",
86
- "training_sources": {
87
- "turn-load": {
88
- "trainer_commit": "3d79a91c92",
89
- "factory_commit": "bf0ee33",
90
- "training_rows_sha256": "2ec8ce650fd19b0e22bbc6b1673be699165fa258e98bbf5f963e6c9354ba80bd",
91
- "validation_rows_sha256": "67c4c86e3385f0241048f2633d97a4aad78a53a3b7a540592937a20eba669fd3",
92
- "corpus_sha256": "c77abd8b40965c05cc43a7545e8c2db2405bdcfc18a73bad254632d523b67bcd"
93
- },
94
- "unit-rank": {
95
- "trainer_commit": "946aea8506c9440a662f87a8d5479963f27735d4",
96
- "training_factory_commit": "b60c15122b4dcddaee75bc45b2e9413a0fe3fa8e",
97
- "evaluation_factory_commit": "5f2b0b0751030dba9fe45d08f3c2e40f27cab4f9"
98
- },
99
- "code-rank": {
100
- "status": "Selected open1 weights retained; training metadata is embedded in the head. No standalone code-rank benchmark report is asserted by this bundle."
101
- }
102
  },
103
- "source_availability": "The separate companion dataset contains training inputs, sanitized conversation exports, trainer snapshots and provenance. Private originals and additional operational artifacts remain archived separately. Historical commits are not promised to be reachable in a flattened repository.",
104
- "evidence": [
105
- "evidence/tool-validation.json",
106
- "evidence/unit-rank-historical-acceptance.json",
107
- "evidence/host-policy-context.json"
108
- ],
109
- "replay_compatibility": "Evidence retains the original scorer semantics. It is not the legacy replay_eval overall schema consumed by bialy audit heads.",
110
- "assembly": "Copied selected weights without modification. Public reports are aggregate projections of existing evidence; no training, inference, or tests were run during packaging.",
111
- "public_name": "Bialy",
112
- "package_version": "open1-b5-e4-bialy",
113
- "dataset_repo": "paintedwolfcode/bialy-dataset"
114
  }
 
1
  {
2
+ "version": "open1-b5-b7g-e4",
3
+ "dataset_version": "open1-b7g-e4",
4
+ "engine_commit": "68e19e34e7b85a459a2a007af52f3611d6996089",
 
 
 
 
 
5
  "heads": [
6
+ {
7
+ "file": "code-rank.safetensors",
8
+ "sha256": "abe235c347a223e957a0f8b7b979e3a584aebab76ee239e5f62117e839b9c40a",
9
+ "metadata": {
10
+ "label": "open1-code-rank",
11
+ "model": "convaiinnovations/laya-multilingual",
12
+ "val_loss": "0.734294557281827",
13
+ "train_examples": "251360",
14
+ "backbone": "jhu-clsp/mmBERT-base",
15
+ "max_len": "512",
16
+ "format": "pw-decide-head/1",
17
+ "val_acc": "0.684346701164295",
18
+ "val_mae": "0.7976714370407317"
19
+ }
20
+ },
21
+ {
22
+ "file": "guide-load.safetensors",
23
+ "sha256": "e0305aa8b62ddc144774785f2e0d3710604af65cd21015662428870f8d34b76e",
24
+ "metadata": {
25
+ "tool_truth": "consensus",
26
+ "families": "guides",
27
+ "rank_levels": "blended",
28
+ "seed": "11",
29
+ "tool_weight": "none",
30
+ "format": "pw-decide-head/1",
31
+ "head_max_len": "512",
32
+ "pos_weight": "6.0",
33
+ "by_type": "{\"loss\": {\"guides\": 0.9665}, \"guides\": {\"precision\": 0.609, \"recall\": 0.947, \"positives\": 1811, \"options\": 4228}, \"guides@coordinator\": {\"precision\": 0.609, \"recall\": 0.947, \"positives\": 1593, \"options\": 3699}, \"guides@worker\": {\"precision\": 0.612, \"recall\": 0.95, \"positives\": 218, \"options\": 529}}",
34
+ "train_rows": "2746",
35
+ "backbone": "jhu-clsp/mmBERT-base",
36
+ "label": "open1-turn-load-B7G-release-independent",
37
+ "val_loss": "0.966544765735868",
38
+ "val_acc": "0.7171239356669821",
39
+ "state_tokens": "{\"guides\": {\"median\": 84, \"under_16\": 0.0, \"request_kept\": 1.0}}",
40
+ "model": "convaiinnovations/laya-multilingual",
41
+ "corpus": "dc3bf9215aac8877",
42
+ "tool_encoding": "independent",
43
+ "turn_rows": "engine-off",
44
+ "tool_option_words": "60",
45
+ "holdout": "[]",
46
+ "tool_negatives": "24",
47
+ "max_len": "1024"
48
+ }
49
+ },
50
  {
51
  "file": "turn-load.safetensors",
52
  "sha256": "550f30b94d5c6a8680d2cb2a607ebc9ee33e709c7679f72606d1e9fda17f2e90",
 
53
  "metadata": {
54
  "by_type": "{\"loss\": {\"tools\": 0.0732}, \"tools\": {\"precision\": 0.417, \"recall\": 0.691, \"positives\": 285, \"options\": 30083}, \"tools@coordinator\": {\"precision\": 0.422, \"recall\": 0.691, \"positives\": 285, \"options\": 27810}, \"tools@worker\": {\"precision\": 0.0, \"recall\": 0.0, \"positives\": 0, \"options\": 2273}}",
55
  "head_max_len": "512",
 
78
  {
79
  "file": "unit-rank.safetensors",
80
  "sha256": "c62674993eabba94660eccfafbb6e04d1549adc7335656c0ddefa7f068fa58f7",
 
81
  "metadata": {
82
  "val_loss": "0.58553760010303",
83
  "format": "pw-decide-head/1",
 
99
  "rank_levels": "skills-blended",
100
  "state_tokens": "{}"
101
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
102
  }
103
  ],
104
+ "evals": [],
105
+ "baseline": {
106
+ "label": "base checkpoint",
107
+ "sets": []
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
108
  },
109
+ "rerank": []
 
 
 
 
 
 
 
 
 
 
110
  }
README.md CHANGED
@@ -1,115 +1,57 @@
1
  ---
2
  license: apache-2.0
3
  base_model: convaiinnovations/laya-multilingual
4
- tags:
5
- - safetensors
6
- - coding-agents
7
- - tool-selection
8
- - decision-heads
9
  ---
10
 
11
- # Bialy
12
 
13
- Package `open1-b5-e4-bialy` preserves model release `open1-b5-e4` and contains three trained decision heads used by Painted Wolf Code:
 
 
 
 
 
 
 
 
14
 
15
- - `turn-load`: B5 independently scores candidate tools for preloading.
16
- - `unit-rank`: E4 dense1 scores skills and requested tools for relevance.
17
- - `code-rank`: open1 scores candidate code units for retrieval.
 
 
 
18
 
19
- These are head weights, not a standalone language model. They require the
20
- Laya multilingual backbone and the compatible `bialy` runtime. This repository
21
- does not bundle the backbone. Training inputs and sanitized session exports are
22
- prepared separately in the companion dataset `paintedwolfcode/bialy-dataset`.
23
- Preparation does not assert that either bundle has been published.
24
 
25
- ## Files and identity
 
 
 
26
 
27
- | File | Label | SHA-256 |
28
- |---|---|---|
29
- | `turn-load.safetensors` | `open1-turn-load-B5-release-independent` | `550f30b94d5c6a8680d2cb2a607ebc9ee33e709c7679f72606d1e9fda17f2e90` |
30
- | `unit-rank.safetensors` | `open1-unit-rank-e4-dense1` | `c62674993eabba94660eccfafbb6e04d1549adc7335656c0ddefa7f068fa58f7` |
31
- | `code-rank.safetensors` | `open1-code-rank` | `abe235c347a223e957a0f8b7b979e3a584aebab76ee239e5f62117e839b9c40a` |
32
 
33
- `release.json` pins the weights, backbone revision, and tool option vocabulary.
34
- `PROVENANCE.json` records the available training identities and evidence scope.
35
- `SHA256SUMS` covers every public file; the source repository's
36
- `releases/heads-open1-b5-e4-bialy.json` anchors that checksum list.
 
37
 
38
- ## Loading the heads
 
39
 
40
- Use the Painted Wolf Code runtime and the backbone
41
- `convaiinnovations/laya-multilingual` at revision
42
- `e4e9ddf21a7b1903b7acffd8814ad4307bf63a67`. Its encoder is `jhu-clsp/mmBERT-base`.
43
 
44
- When building Painted Wolf Code, set `BIALY_HEADS_DIR` to the directory containing
45
- these three `.safetensors` files. The app's committed decision release manifest
46
- must match their hashes; the staging process verifies them before packaging.
47
- The B5 head uses independent tool encoding, a 1024-token maximum length, and a
48
- 512-token head budget. A generic Transformers pipeline is not provided.
49
 
50
- ## Tool validation
51
 
52
- The existing validation set contains 445 complete turns. Consensus labels and
53
- observed tool calls are separate evaluation targets. B2 is the previous trained
54
- tool head, not an untuned backbone baseline. B5 selection used this validation
55
- distribution; these results are not fresh final acceptance or end-to-end task success.
56
-
57
- This compares two complete configurations: B2 uses joint encoding with six-word
58
- tool options; B5 uses independent scoring with up to 60-word options. Their corpus
59
- hashes intentionally differ. Each candidate was bound to its own corpus, prediction
60
- metadata, head hash and predictions; their validation-row hash is shared.
61
-
62
- | Head | Cutoff | Consensus precision | Consensus recall | Observed-call precision | Observed-call recall | Loads per turn |
63
- |---|---|---|---|---|---|---|
64
- | B2 | 0.89 | 0.757 | 0.295 | 0.517 | 0.149 | 0.339 |
65
- | B5 | 0.95 | 0.822 | 0.291 | 0.547 | 0.143 | 0.308 |
66
- | B5 | 0.94 | 0.798 | 0.319 | 0.559 | 0.172 | 0.362 |
67
-
68
- The 0.94 and 0.95 rows describe two evaluated cutoffs for the same B5 weights.
69
- The archived host closeout uses tool preload 0.94, at most one automatic skill at
70
- 3.4, and free-text skill lookup 3.2. Those settings do not change the weights.
71
- Later host tests and startup checks establish implementation correctness within
72
- their recorded scope, not model task quality or fresh end-to-end acceptance.
73
- App policy controls cutoffs independently of head files. The evidence file
74
- retains all archived aggregate threshold rows, input hashes, and report hashes.
75
- No benchmark was rerun for this publication bundle.
76
-
77
- ## Historical unit-rank evidence
78
-
79
- E4's archived acceptance report covers 789 requests, of which 660 have a relevant
80
- skill. Under its historical listing and preload policy, a relevant skill appeared
81
- in the first six for 92.1% of relevant requests; 122 of 123 automatic preloads were
82
- relevant. The report contains no observed request-tool needs.
83
-
84
- The strict skills verdict was **failed**: the common-skill criterion did not pass.
85
- The other listed skill criteria passed. This evidence does not establish acceptance
86
- for current free-text lookup or current
87
- automatic skill preloading. The original policy values and verdict booleans are
88
- included in `evidence/unit-rank-historical-acceptance.json`.
89
-
90
- ## Limits and evidence access
91
-
92
- Only aggregate evaluation evidence is published here. Per-example requests,
93
- training rows, session stores, raw prediction text, machine paths, and operational
94
- logs are excluded. Original report and input hashes preserve evidence identities;
95
- they do not make omitted inputs publicly reproducible. Inputs, source snapshots,
96
- predictions and training artifacts are archived privately. That private archive
97
- is distinct from what a reader can reproduce from this weights-only public bundle.
98
- The companion dataset package `open1-b5-e4-bialy` provides training inputs,
99
- sanitized conversation exports and historical trainer snapshots separately.
100
- Historical source commits are recorded without claiming they are reachable in
101
- a flattened public repository.
102
- No standalone code-rank benchmark or full-current-policy acceptance claim is made.
103
-
104
- `bialy verify heads` can check this bundle against its committed checksum anchor.
105
- The included reports use the tool-validation and ranking scorer formats, so the
106
- legacy `bialy audit heads` replay interface does not support this evidence bundle.
107
- It should not be presented as a successful replay audit.
108
-
109
- ## Attribution
110
-
111
- The heads use [Laya](https://github.com/NandhaKishorM/laya), by Convai Innovations,
112
- and its [multilingual checkpoint](https://huggingface.co/convaiinnovations/laya-multilingual),
113
- over [mmBERT-base](https://huggingface.co/jhu-clsp/mmBERT-base).
114
- Head weights are provided under Apache-2.0. See `LICENSE` and `NOTICE` for
115
- attribution and the separately licensed backbone dependencies.
 
1
  ---
2
  license: apache-2.0
3
  base_model: convaiinnovations/laya-multilingual
4
+ datasets: [paintedwolfcode/bialy-dataset]
5
+ tags: [coding-agents, tool-selection]
 
 
 
6
  ---
7
 
8
+ # Bialy decision heads open1-b5-b7g-e4
9
 
10
+ Tuned heads built on [Laya](https://github.com/NandhaKishorM/laya) by Convai Innovations,
11
+ over its frozen multilingual encoder that Painted Wolf Code
12
+ asks as it works: which loadable tool schemas a turn will need, which
13
+ instruction units it can leave out, and what kind of work it is (`turn-load`);
14
+ how relevant each skill and tool card is to a request (`unit-rank`); and which
15
+ code units best answer a request, blended into the text-match order of code
16
+ summaries, repository maps, and project search (`code-rank`). The engine
17
+ (`bialy`) loads them beside the backbone; a head trained over another
18
+ backbone is refused.
19
 
20
+ Trained on [paintedwolfcode/bialy-dataset](https://huggingface.co/datasets/paintedwolfcode/bialy-dataset) open1-b7g-e4: `turn-load`
21
+ and `unit-rank` on its session rows, labeled by what open-weights models did
22
+ in coding-agent sessions and scored by an open-weights judge; `code-rank` on its
23
+ code-rank pairs, requests open-weights models wrote for code units in the same
24
+ repositories. Trained and replayed with Painted Wolf Code at commit
25
+ 68e19e34e7b85a459a2a007af52f3611d6996089; packaged with https://github.com/paintedwolf-ai/bialy.
26
 
27
+ ## Heads
 
 
 
 
28
 
29
+ - `code-rank.safetensors`: open1-code-rank, over jhu-clsp/mmBERT-base, sha256 `abe235c347a223e957a0f8b7b979e3a584aebab76ee239e5f62117e839b9c40a`
30
+ - `guide-load.safetensors`: open1-turn-load-B7G-release-independent, over jhu-clsp/mmBERT-base, sha256 `e0305aa8b62ddc144774785f2e0d3710604af65cd21015662428870f8d34b76e`
31
+ - `turn-load.safetensors`: open1-turn-load-B5-release-independent, over jhu-clsp/mmBERT-base, sha256 `550f30b94d5c6a8680d2cb2a607ebc9ee33e709c7679f72606d1e9fda17f2e90`
32
+ - `unit-rank.safetensors`: open1-unit-rank-e4-dense1, over jhu-clsp/mmBERT-base, sha256 `c62674993eabba94660eccfafbb6e04d1549adc7335656c0ddefa7f068fa58f7`
33
 
34
+ ## Results
 
 
 
 
35
 
36
+ Replayed through the shipped engine on sets neither head trained on. Tool
37
+ columns are precision / recall / F1 of the loadable tools a turn used, at
38
+ the catalog's load threshold; guide omission precision is the share of
39
+ omitted instruction units the turn did not need; need MRR ranks the tools a
40
+ `request_tools` need went on to use.
41
 
42
+ | Set | Heads | tools P/R/F1 | macro R | loads/turn | guide omit P | kind acc | need MRR |
43
+ |---|---|---|---|---|---|---|---|
44
 
45
+ `code-rank`, on repositories it never trained on: the rank of the code unit a
46
+ request was written for, in the site's text-match order and blended with the
47
+ head.
48
 
49
+ | Report | Pairs | MRR text / blended | hit@1 text / blended | improved / regressed |
50
+ |---|---|---|---|---|
 
 
 
51
 
52
+ ## Check it yourself
53
 
54
+ Each row of the table comes from a replay report shipped in `eval/`.
55
+ `bialy audit heads` (from https://github.com/paintedwolf-ai/bialy) replays these heads through the
56
+ Painted Wolf Code engine on a CPU over the dataset's held-out split and
57
+ compares every metric with the shipped report.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
SHA256SUMS CHANGED
@@ -1,11 +1,8 @@
1
  3c2307bbbbb88f79560a165b098c600da8f20fb1ec024586d4be192ebc9c6f8e LICENSE
2
  b4cdc8ad8e0ec0cc6def7b0c1ad077e5bbd8f473f0ba8bc6a66dd5a46a71abee NOTICE
3
- ce81c8201ed2afb855c95d99d70ccc107ec8577937ea98ef3506e29eb1d520a6 PROVENANCE.json
4
- d6001f48ac599ef2361a0b1026bf2780b62c6565dab2eb1f7c24b8c05e3988fd README.md
5
  abe235c347a223e957a0f8b7b979e3a584aebab76ee239e5f62117e839b9c40a code-rank.safetensors
6
- 0f338ab310d2746541ca08799fcbb76ab25dbdc724a8e41731ea874091ff0549 evidence/host-policy-context.json
7
- 45ea0d69658a114dc73f3984766e529c66a87f6f35fc2deb536d26b84313cd9d evidence/tool-validation.json
8
- 5b97c4f57274870cdcd11a2912baa83c69dc8e40f74df9409f628c076d89a20e evidence/unit-rank-historical-acceptance.json
9
- 404f04b03fcf6b757330eda9c0841835248b1f8dc105fb3b61f01dfa9657c872 release.json
10
  550f30b94d5c6a8680d2cb2a607ebc9ee33e709c7679f72606d1e9fda17f2e90 turn-load.safetensors
11
  c62674993eabba94660eccfafbb6e04d1549adc7335656c0ddefa7f068fa58f7 unit-rank.safetensors
 
1
  3c2307bbbbb88f79560a165b098c600da8f20fb1ec024586d4be192ebc9c6f8e LICENSE
2
  b4cdc8ad8e0ec0cc6def7b0c1ad077e5bbd8f473f0ba8bc6a66dd5a46a71abee NOTICE
3
+ a364297432e17a7a1fec7a0bf38abea1c787c858e267c1828d0227451c664335 PROVENANCE.json
4
+ 2eeca00baf6bd1fc21e23c42b759251b355bfcdbd12b0cfa16015534449e110d README.md
5
  abe235c347a223e957a0f8b7b979e3a584aebab76ee239e5f62117e839b9c40a code-rank.safetensors
6
+ e0305aa8b62ddc144774785f2e0d3710604af65cd21015662428870f8d34b76e guide-load.safetensors
 
 
 
7
  550f30b94d5c6a8680d2cb2a607ebc9ee33e709c7679f72606d1e9fda17f2e90 turn-load.safetensors
8
  c62674993eabba94660eccfafbb6e04d1549adc7335656c0ddefa7f068fa58f7 unit-rank.safetensors
guide-load.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e0305aa8b62ddc144774785f2e0d3710604af65cd21015662428870f8d34b76e
3
+ size 59881676