Instructions to use kinit/hyperproof-solver-sft with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use kinit/hyperproof-solver-sft with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("kinit/hyperproof-solver-sft", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download checkpoint-48/trainer_state.json from kinit/hyperproof-solver-sft: direct link, hf CLI and curl.
- Browser
- Download file 3.53 kB
-
https://huggingface.co/kinit/hyperproof-solver-sft/resolve/main/checkpoint-48/trainer_state.json
- Command line
-
hf download hf://kinit/hyperproof-solver-sft/checkpoint-48/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/kinit/hyperproof-solver-sft/resolve/main/checkpoint-48/trainer_state.json
3.53 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.7619047619047619, | |
| "eval_steps": 500, | |
| "global_step": 48, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "entropy": 0.25798711776733396, | |
| "epoch": 0.07936507936507936, | |
| "grad_norm": 0.6395628452301025, | |
| "learning_rate": 3.3333333333333335e-05, | |
| "loss": 0.33170533180236816, | |
| "mean_token_accuracy": 0.9083019256591797, | |
| "num_tokens": 2249744.0, | |
| "step": 5 | |
| }, | |
| { | |
| "entropy": 0.1518770933151245, | |
| "epoch": 0.15873015873015872, | |
| "grad_norm": 0.27742329239845276, | |
| "learning_rate": 4.9966852247120764e-05, | |
| "loss": 0.14515713453292847, | |
| "mean_token_accuracy": 0.9476617932319641, | |
| "num_tokens": 4502357.0, | |
| "step": 10 | |
| }, | |
| { | |
| "entropy": 0.1205204889178276, | |
| "epoch": 0.23809523809523808, | |
| "grad_norm": 0.15280286967754364, | |
| "learning_rate": 4.976460088644493e-05, | |
| "loss": 0.11978392601013184, | |
| "mean_token_accuracy": 0.9552689790725708, | |
| "num_tokens": 6750630.0, | |
| "step": 15 | |
| }, | |
| { | |
| "entropy": 0.1094709500670433, | |
| "epoch": 0.31746031746031744, | |
| "grad_norm": 0.13621626794338226, | |
| "learning_rate": 4.938000100611456e-05, | |
| "loss": 0.10746543407440186, | |
| "mean_token_accuracy": 0.9598726153373718, | |
| "num_tokens": 8991471.0, | |
| "step": 20 | |
| }, | |
| { | |
| "entropy": 0.10419090241193771, | |
| "epoch": 0.3968253968253968, | |
| "grad_norm": 0.16320250928401947, | |
| "learning_rate": 4.881588452008456e-05, | |
| "loss": 0.10661067962646484, | |
| "mean_token_accuracy": 0.9599695205688477, | |
| "num_tokens": 11228489.0, | |
| "step": 25 | |
| }, | |
| { | |
| "entropy": 0.09569432139396668, | |
| "epoch": 0.47619047619047616, | |
| "grad_norm": 0.15873780846595764, | |
| "learning_rate": 4.80764051721044e-05, | |
| "loss": 0.09594419002532958, | |
| "mean_token_accuracy": 0.9637031435966492, | |
| "num_tokens": 13464387.0, | |
| "step": 30 | |
| }, | |
| { | |
| "entropy": 0.09625398516654968, | |
| "epoch": 0.5555555555555556, | |
| "grad_norm": 0.15297284722328186, | |
| "learning_rate": 4.7167007950568505e-05, | |
| "loss": 0.09495124220848083, | |
| "mean_token_accuracy": 0.964220929145813, | |
| "num_tokens": 15709681.0, | |
| "step": 35 | |
| }, | |
| { | |
| "entropy": 0.09195081293582916, | |
| "epoch": 0.6349206349206349, | |
| "grad_norm": 0.15219081938266754, | |
| "learning_rate": 4.609438899557964e-05, | |
| "loss": 0.09150891304016114, | |
| "mean_token_accuracy": 0.9652478575706482, | |
| "num_tokens": 17987403.0, | |
| "step": 40 | |
| }, | |
| { | |
| "entropy": 0.08980809897184372, | |
| "epoch": 0.7142857142857143, | |
| "grad_norm": 0.1412411779165268, | |
| "learning_rate": 4.4866446293440626e-05, | |
| "loss": 0.0898817539215088, | |
| "mean_token_accuracy": 0.965602707862854, | |
| "num_tokens": 20231746.0, | |
| "step": 45 | |
| } | |
| ], | |
| "logging_steps": 5, | |
| "max_steps": 189, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 3, | |
| "save_steps": 8, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.2036545538078802e+18, | |
| "train_batch_size": 4, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |