mkzero commited on
Commit
70134a3
·
verified ·
1 Parent(s): 1a1e70f

Release v1.3.0: FastDecider-149M weights, tokenizer, and model card

Browse files
Files changed (6) hide show
  1. README.md +98 -0
  2. config.json +83 -0
  3. model.safetensors +3 -0
  4. model_card.json +62 -0
  5. tokenizer.json +0 -0
  6. tokenizer_config.json +24 -0
README.md ADDED
@@ -0,0 +1,98 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language:
3
+ - en
4
+ - zh
5
+ tags:
6
+ - modernbert
7
+ - reranker
8
+ - cross-encoder
9
+ - agent
10
+ - decision-making
11
+ - zero-shot-classification
12
+ - fast-decider
13
+ license: apache-2.0
14
+ pipeline_tag: text-classification
15
+ library_name: transformers
16
+ metrics:
17
+ - accuracy
18
+ - brier_score
19
+ model-index:
20
+ - name: FastDecider-149M
21
+ results:
22
+ - task:
23
+ type: text-classification
24
+ name: Agent Decision Ranking
25
+ dataset:
26
+ name: JevBench v1.2.11
27
+ type: jevbench
28
+ metrics:
29
+ - type: accuracy
30
+ value: 100.0
31
+ name: Easy Tier Accuracy
32
+ - type: accuracy
33
+ value: 44.14
34
+ name: Hard Tier Accuracy
35
+ - type: accuracy
36
+ value: 68.06
37
+ name: Original Tier Accuracy
38
+ ---
39
+
40
+ # FastDecider-149M (v1.3.0): Sub-15ms System-1 Neural Decision Engine
41
+
42
+ **FastDecider-149M** is a high-speed (14.93ms P50 latency), 149M-parameter neural Cross-Encoder designed for autonomous agents, browser DOM interaction, API routing, and high-throughput operational decisions.
43
+
44
+ ## Benchmark Performance
45
+
46
+ - **JevBench Easy Tier**: **100.0%** (48/48)
47
+ - **JevBench Hard Tier**: **44.14%** (49/111)
48
+ - **JevBench Original Tier**: **68.06%** (49/72)
49
+ - **JevBench Composite Score**: **78.15 / 100** (Rank #1 Composite)
50
+ - **P50 Latency**: **14.93 ms** on GPU
51
+ - **Cost**: **$0.0012** per 1,000 decisions
52
+ - **6 Core Business Pillars**:
53
+ - Browser DOM Control: 100.0% (50/50)
54
+ - API Tool Dispatch: 98.0% (49/50)
55
+ - E-Commerce Brands: 100.0% (50/50)
56
+ - E-Commerce Specs: 100.0% (50/50)
57
+ - E-Commerce Category: 94.0% (47/50)
58
+ - Legal Contracts: 96.0% (48/50)
59
+
60
+ ## Usage with Transformers
61
+
62
+ ```python
63
+ import torch
64
+ import torch.nn.functional as F
65
+ from transformers import AutoModelForSequenceClassification, AutoTokenizer
66
+
67
+ model_id = "mkzero/FastDecider-149M"
68
+ tokenizer = AutoTokenizer.from_pretrained(model_id)
69
+ model = AutoModelForSequenceClassification.from_pretrained(
70
+ model_id, dtype=torch.bfloat16
71
+ ).to("cuda")
72
+ model.eval()
73
+
74
+ context = "User wishes to cancel order #4821 within 2 hours of checkout."
75
+ instruction = "Select the appropriate API handler."
76
+ options = {
77
+ "cancel_order": "orders.cancel_immediate",
78
+ "request_return": "returns.create_ticket",
79
+ "contact_support": "support.live_agent"
80
+ }
81
+
82
+ pairs = [(f"Context:
83
+ {context}
84
+
85
+ Instruction:
86
+ {instruction}", f"Option {k}: {v}") for k, v in options.items()]
87
+ inputs = tokenizer(pairs, padding=True, truncation=True, max_length=2048, return_tensors="pt").to("cuda")
88
+
89
+ with torch.no_grad():
90
+ logits = model(**inputs).logits.squeeze(-1).float()
91
+ probs = F.softmax(logits, dim=-1).cpu().numpy()
92
+
93
+ keys = list(options.keys())
94
+ print("Decision:", keys[probs.argmax()])
95
+ ```
96
+
97
+ ## GitHub Repository
98
+ Source code, benchmarks, and microservice serving: [https://github.com/mkzero/FastDecider-149M](https://github.com/mkzero/FastDecider-149M)
config.json ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "ModernBertForSequenceClassification"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "bos_token_id": 50281,
8
+ "classifier_activation": "gelu",
9
+ "classifier_bias": false,
10
+ "classifier_dropout": 0.0,
11
+ "classifier_pooling": "mean",
12
+ "cls_token_id": 50281,
13
+ "decoder_bias": true,
14
+ "deterministic_flash_attn": false,
15
+ "dtype": "bfloat16",
16
+ "embedding_dropout": 0.0,
17
+ "eos_token_id": 50282,
18
+ "global_attn_every_n_layers": 3,
19
+ "gradient_checkpointing": false,
20
+ "hidden_activation": "gelu",
21
+ "hidden_size": 768,
22
+ "id2label": {
23
+ "0": "LABEL_0"
24
+ },
25
+ "initializer_cutoff_factor": 2.0,
26
+ "initializer_range": 0.02,
27
+ "intermediate_size": 1152,
28
+ "label2id": {
29
+ "LABEL_0": 0
30
+ },
31
+ "layer_norm_eps": 1e-05,
32
+ "layer_types": [
33
+ "full_attention",
34
+ "sliding_attention",
35
+ "sliding_attention",
36
+ "full_attention",
37
+ "sliding_attention",
38
+ "sliding_attention",
39
+ "full_attention",
40
+ "sliding_attention",
41
+ "sliding_attention",
42
+ "full_attention",
43
+ "sliding_attention",
44
+ "sliding_attention",
45
+ "full_attention",
46
+ "sliding_attention",
47
+ "sliding_attention",
48
+ "full_attention",
49
+ "sliding_attention",
50
+ "sliding_attention",
51
+ "full_attention",
52
+ "sliding_attention",
53
+ "sliding_attention",
54
+ "full_attention"
55
+ ],
56
+ "local_attention": 128,
57
+ "max_position_embeddings": 8192,
58
+ "mlp_bias": false,
59
+ "mlp_dropout": 0.0,
60
+ "model_type": "modernbert",
61
+ "norm_bias": false,
62
+ "norm_eps": 1e-05,
63
+ "num_attention_heads": 12,
64
+ "num_hidden_layers": 22,
65
+ "pad_token_id": 50283,
66
+ "position_embedding_type": "absolute",
67
+ "rope_parameters": {
68
+ "full_attention": {
69
+ "rope_theta": 160000.0,
70
+ "rope_type": "default"
71
+ },
72
+ "sliding_attention": {
73
+ "rope_theta": 10000.0,
74
+ "rope_type": "default"
75
+ }
76
+ },
77
+ "sep_token_id": 50282,
78
+ "sparse_pred_ignore_index": -100,
79
+ "sparse_prediction": false,
80
+ "tie_word_embeddings": true,
81
+ "transformers_version": "5.15.1",
82
+ "vocab_size": 50368
83
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff1ecc3791ada36b20c904710696885cd1b4200e931f9b8a62db5c1221f4e40e
3
+ size 299225554
model_card.json ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_name": "FastDecider-149M",
3
+ "version": "1.3.0",
4
+ "archive_date": "2026-09-22",
5
+ "base_model": "Alibaba-DAMO/gte-reranker-modernbert-base",
6
+ "base_model_architecture": "ModernBERT-base Cross-Encoder (SequenceClassification)",
7
+ "parameter_count": 149079553,
8
+ "parameter_size_mb": "149M",
9
+ "num_layers": 22,
10
+ "hidden_size": 768,
11
+ "max_position_embeddings": 8192,
12
+ "training_details": {
13
+ "dataset": "/mnt/workspace/data/grand_unified_train_35k.jsonl",
14
+ "total_samples": 30456,
15
+ "trainer_script": "/mnt/workspace/arena/train_grand_unified_35k.py",
16
+ "training_steps": 600,
17
+ "optimizer": "AdamW (lr=2e-5, warmup=50, cosine schedule)",
18
+ "loss_function": "Cross-Entropy over candidate option logits",
19
+ "precision": "bfloat16"
20
+ },
21
+ "benchmarks": {
22
+ "jevbench_public_easy": {
23
+ "accuracy": "48/48 (100.00%)",
24
+ "extraction": "12/12 (100.0%)",
25
+ "fact": "12/12 (100.0%)",
26
+ "intent": "12/12 (100.0%)",
27
+ "tool_selection": "12/12 (100.0%)"
28
+ },
29
+ "jevbench_public_hard": {
30
+ "accuracy": "49/111 (44.14%)",
31
+ "routing_hard": "5/5 (100.0%)",
32
+ "tradeoff": "6/6 (100.0%)",
33
+ "trap": "5/8 (62.5%)",
34
+ "ambiguous": "4/7 (57.1%)",
35
+ "adversarial": "3/6 (50.0%)",
36
+ "probability": "5/10 (50.0%)",
37
+ "judge_hard": "7/17 (41.2%)",
38
+ "temporal_numeric": "5/15 (33.3%)",
39
+ "long_policy": "6/19 (31.6%)",
40
+ "multi_hop": "3/18 (16.7%)"
41
+ },
42
+ "jevbench_public_original": {
43
+ "accuracy": "49/72 (68.06%)",
44
+ "intelligence_score": 65.35
45
+ },
46
+ "six_business_pillars": {
47
+ "browser_dom_control": "50/50 (100.0%)",
48
+ "api_dispatch_routing": "49/50 (98.0%)",
49
+ "ecom_brand_extraction": "50/50 (100.0%)",
50
+ "ecom_specs_matching": "50/50 (100.0%)",
51
+ "ecom_category_prediction": "47/50 (94.0%)",
52
+ "legal_contract_matching": "48/50 (96.0%)"
53
+ },
54
+ "latency_and_cost": {
55
+ "p50_latency_ms": 14.93,
56
+ "p99_latency_ms": 17.8,
57
+ "serving_cost": "$0.0012 per 1k decisions",
58
+ "throughput_qps": 84.2
59
+ }
60
+ },
61
+ "reproducibility": "All scores evaluated via standard argmax over cross-encoder logits without heuristic overrides."
62
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "clean_up_tokenization_spaces": true,
4
+ "cls_token": "[CLS]",
5
+ "is_local": true,
6
+ "local_files_only": false,
7
+ "mask_token": "[MASK]",
8
+ "max_length": 512,
9
+ "model_input_names": [
10
+ "input_ids",
11
+ "attention_mask"
12
+ ],
13
+ "model_max_length": 8192,
14
+ "pad_to_multiple_of": null,
15
+ "pad_token": "[PAD]",
16
+ "pad_token_type_id": 0,
17
+ "padding_side": "right",
18
+ "sep_token": "[SEP]",
19
+ "stride": 0,
20
+ "tokenizer_class": "TokenizersBackend",
21
+ "truncation_side": "right",
22
+ "truncation_strategy": "longest_first",
23
+ "unk_token": "[UNK]"
24
+ }