N8Programs commited on 3 days ago

Commit

083750a

verified ·

1 Parent(s): 451a228

Add NextTerm tokenizer files and MLX config compatibility

Browse files

Files changed (24) hide show

checkpoints/best_val/config.json +2 -1
checkpoints/best_val/special_tokens_map.json +6 -0
checkpoints/best_val/tokenizer.json +149 -0
checkpoints/best_val/tokenizer_config.json +36 -0
checkpoints/checkpoint_tokens_012000258345/config.json +2 -1
checkpoints/checkpoint_tokens_012000258345/special_tokens_map.json +6 -0
checkpoints/checkpoint_tokens_012000258345/tokenizer.json +149 -0
checkpoints/checkpoint_tokens_012000258345/tokenizer_config.json +36 -0
checkpoints/checkpoint_tokens_012500265837/config.json +2 -1
checkpoints/checkpoint_tokens_012500265837/special_tokens_map.json +6 -0
checkpoints/checkpoint_tokens_012500265837/tokenizer.json +149 -0
checkpoints/checkpoint_tokens_012500265837/tokenizer_config.json +36 -0
checkpoints/checkpoint_tokens_013000266889/config.json +2 -1
checkpoints/checkpoint_tokens_013000266889/special_tokens_map.json +6 -0
checkpoints/checkpoint_tokens_013000266889/tokenizer.json +149 -0
checkpoints/checkpoint_tokens_013000266889/tokenizer_config.json +36 -0
checkpoints/checkpoint_tokens_013500289737/config.json +2 -1
checkpoints/checkpoint_tokens_013500289737/special_tokens_map.json +6 -0
checkpoints/checkpoint_tokens_013500289737/tokenizer.json +149 -0
checkpoints/checkpoint_tokens_013500289737/tokenizer_config.json +36 -0
checkpoints/final_latest/config.json +2 -1
checkpoints/final_latest/special_tokens_map.json +6 -0
checkpoints/final_latest/tokenizer.json +149 -0
checkpoints/final_latest/tokenizer_config.json +36 -0

checkpoints/best_val/config.json CHANGED Viewed

@@ -59,5 +59,6 @@
   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
-  "vocab_size": 16
 }

   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
+  "vocab_size": 16,
+  "rope_theta": 1000000.0
 }

checkpoints/best_val/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<bos>",
+  "eos_token": "<eos>",
+  "pad_token": "<pad>",
+  "unk_token": "<pad>"
+}

checkpoints/best_val/tokenizer.json ADDED Viewed

	@@ -0,0 +1,149 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 12,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "[^0-9,-]+"
+        },
+        "content": ""
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": ",+"
+        },
+        "content": ","
+      },
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "^,+"
+        },
+        "content": ""
+      },
+  { "type": "Replace", "pattern": { "Regex": ",{2,}$" }, "content": "," }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": ""
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          12
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "Sequence",
+    "decoders": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "String": " "
+        },
+        "content": ""
+      }
+    ]
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "0": 0,
+      "1": 1,
+      "2": 2,
+      "3": 3,
+      "4": 4,
+      "5": 5,
+      "6": 6,
+      "7": 7,
+      "8": 8,
+      "9": 9,
+      "-": 10,
+      ",": 11,
+      "<bos>": 12,
+      "<eos>": 13,
+      "<pad>": 14
+    },
+    "unk_token": "<pad>"
+  }
+}

checkpoints/best_val/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "added_tokens_decoder": {
+    "12": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<eos>",
+  "extra_special_tokens": {},
+  "model_max_length": 40960,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_012000258345/config.json CHANGED Viewed

@@ -59,5 +59,6 @@
   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
-  "vocab_size": 16
 }

   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
+  "vocab_size": 16,
+  "rope_theta": 1000000.0
 }

checkpoints/checkpoint_tokens_012000258345/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<bos>",
+  "eos_token": "<eos>",
+  "pad_token": "<pad>",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_012000258345/tokenizer.json ADDED Viewed

	@@ -0,0 +1,149 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 12,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "[^0-9,-]+"
+        },
+        "content": ""
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": ",+"
+        },
+        "content": ","
+      },
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "^,+"
+        },
+        "content": ""
+      },
+  { "type": "Replace", "pattern": { "Regex": ",{2,}$" }, "content": "," }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": ""
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          12
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "Sequence",
+    "decoders": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "String": " "
+        },
+        "content": ""
+      }
+    ]
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "0": 0,
+      "1": 1,
+      "2": 2,
+      "3": 3,
+      "4": 4,
+      "5": 5,
+      "6": 6,
+      "7": 7,
+      "8": 8,
+      "9": 9,
+      "-": 10,
+      ",": 11,
+      "<bos>": 12,
+      "<eos>": 13,
+      "<pad>": 14
+    },
+    "unk_token": "<pad>"
+  }
+}

checkpoints/checkpoint_tokens_012000258345/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "added_tokens_decoder": {
+    "12": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<eos>",
+  "extra_special_tokens": {},
+  "model_max_length": 40960,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_012500265837/config.json CHANGED Viewed

@@ -59,5 +59,6 @@
   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
-  "vocab_size": 16
 }

   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
+  "vocab_size": 16,
+  "rope_theta": 1000000.0
 }

checkpoints/checkpoint_tokens_012500265837/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<bos>",
+  "eos_token": "<eos>",
+  "pad_token": "<pad>",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_012500265837/tokenizer.json ADDED Viewed

	@@ -0,0 +1,149 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 12,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "[^0-9,-]+"
+        },
+        "content": ""
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": ",+"
+        },
+        "content": ","
+      },
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "^,+"
+        },
+        "content": ""
+      },
+  { "type": "Replace", "pattern": { "Regex": ",{2,}$" }, "content": "," }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": ""
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          12
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "Sequence",
+    "decoders": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "String": " "
+        },
+        "content": ""
+      }
+    ]
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "0": 0,
+      "1": 1,
+      "2": 2,
+      "3": 3,
+      "4": 4,
+      "5": 5,
+      "6": 6,
+      "7": 7,
+      "8": 8,
+      "9": 9,
+      "-": 10,
+      ",": 11,
+      "<bos>": 12,
+      "<eos>": 13,
+      "<pad>": 14
+    },
+    "unk_token": "<pad>"
+  }
+}

checkpoints/checkpoint_tokens_012500265837/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "added_tokens_decoder": {
+    "12": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<eos>",
+  "extra_special_tokens": {},
+  "model_max_length": 40960,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_013000266889/config.json CHANGED Viewed

@@ -59,5 +59,6 @@
   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
-  "vocab_size": 16
 }

   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
+  "vocab_size": 16,
+  "rope_theta": 1000000.0
 }

checkpoints/checkpoint_tokens_013000266889/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<bos>",
+  "eos_token": "<eos>",
+  "pad_token": "<pad>",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_013000266889/tokenizer.json ADDED Viewed

	@@ -0,0 +1,149 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 12,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "[^0-9,-]+"
+        },
+        "content": ""
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": ",+"
+        },
+        "content": ","
+      },
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "^,+"
+        },
+        "content": ""
+      },
+  { "type": "Replace", "pattern": { "Regex": ",{2,}$" }, "content": "," }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": ""
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          12
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "Sequence",
+    "decoders": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "String": " "
+        },
+        "content": ""
+      }
+    ]
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "0": 0,
+      "1": 1,
+      "2": 2,
+      "3": 3,
+      "4": 4,
+      "5": 5,
+      "6": 6,
+      "7": 7,
+      "8": 8,
+      "9": 9,
+      "-": 10,
+      ",": 11,
+      "<bos>": 12,
+      "<eos>": 13,
+      "<pad>": 14
+    },
+    "unk_token": "<pad>"
+  }
+}

checkpoints/checkpoint_tokens_013000266889/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "added_tokens_decoder": {
+    "12": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<eos>",
+  "extra_special_tokens": {},
+  "model_max_length": 40960,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_013500289737/config.json CHANGED Viewed

@@ -59,5 +59,6 @@
   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
-  "vocab_size": 16
 }

   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
+  "vocab_size": 16,
+  "rope_theta": 1000000.0
 }

checkpoints/checkpoint_tokens_013500289737/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<bos>",
+  "eos_token": "<eos>",
+  "pad_token": "<pad>",
+  "unk_token": "<pad>"
+}

checkpoints/checkpoint_tokens_013500289737/tokenizer.json ADDED Viewed

	@@ -0,0 +1,149 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 12,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "[^0-9,-]+"
+        },
+        "content": ""
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": ",+"
+        },
+        "content": ","
+      },
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "^,+"
+        },
+        "content": ""
+      },
+  { "type": "Replace", "pattern": { "Regex": ",{2,}$" }, "content": "," }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": ""
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          12
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "Sequence",
+    "decoders": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "String": " "
+        },
+        "content": ""
+      }
+    ]
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "0": 0,
+      "1": 1,
+      "2": 2,
+      "3": 3,
+      "4": 4,
+      "5": 5,
+      "6": 6,
+      "7": 7,
+      "8": 8,
+      "9": 9,
+      "-": 10,
+      ",": 11,
+      "<bos>": 12,
+      "<eos>": 13,
+      "<pad>": 14
+    },
+    "unk_token": "<pad>"
+  }
+}

checkpoints/checkpoint_tokens_013500289737/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "added_tokens_decoder": {
+    "12": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<eos>",
+  "extra_special_tokens": {},
+  "model_max_length": 40960,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<pad>"
+}

checkpoints/final_latest/config.json CHANGED Viewed

@@ -59,5 +59,6 @@
   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
-  "vocab_size": 16
 }

   "transformers_version": "5.9.0",
   "use_cache": true,
   "use_sliding_window": false,
+  "vocab_size": 16,
+  "rope_theta": 1000000.0
 }

checkpoints/final_latest/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<bos>",
+  "eos_token": "<eos>",
+  "pad_token": "<pad>",
+  "unk_token": "<pad>"
+}

checkpoints/final_latest/tokenizer.json ADDED Viewed

	@@ -0,0 +1,149 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 12,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 13,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 14,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "[^0-9,-]+"
+        },
+        "content": ""
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": ",+"
+        },
+        "content": ","
+      },
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      },
+      {
+        "type": "Replace",
+        "pattern": {
+          "Regex": "^,+"
+        },
+        "content": ""
+      },
+  { "type": "Replace", "pattern": { "Regex": ",{2,}$" }, "content": "," }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": ""
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          12
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "Sequence",
+    "decoders": [
+      {
+        "type": "Replace",
+        "pattern": {
+          "String": " "
+        },
+        "content": ""
+      }
+    ]
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "0": 0,
+      "1": 1,
+      "2": 2,
+      "3": 3,
+      "4": 4,
+      "5": 5,
+      "6": 6,
+      "7": 7,
+      "8": 8,
+      "9": 9,
+      "-": 10,
+      ",": 11,
+      "<bos>": 12,
+      "<eos>": 13,
+      "<pad>": 14
+    },
+    "unk_token": "<pad>"
+  }
+}

checkpoints/final_latest/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,36 @@

+{
+  "added_tokens_decoder": {
+    "12": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<eos>",
+  "extra_special_tokens": {},
+  "model_max_length": 40960,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<pad>"
+}