TensorBoard
Safetensors
English
llama
picollama / continual_state.json
appvoid's picture
Update root model to step 120000
2c7bc38 verified
Raw
History Blame Contribute Delete
1.76 kB
{
"global_step": 120000,
"sequence_tokens_emitted": 23757045760,
"mix": [
0.8,
0.09,
0.05,
0.05,
0.01
],
"fineweb": "HuggingFaceFW/fineweb-edu:default",
"ultra_multi": "openbmb/Ultra-FineWeb-L3:Ultra-FineWeb-L3-en-Multi-Style-Synthetic",
"ultra_qa": "openbmb/Ultra-FineWeb-L3:Ultra-FineWeb-L3-en-QA-Synthetic",
"rewrite": "appvoid/rewrite6",
"noprompt": "appvoid/no-prompt-75k",
"early_mix": [
0.9,
0.05,
0.03,
0.0175,
0.0025
],
"late_mix": [
0.8,
0.09,
0.05,
0.05,
0.01
],
"formatter_version": 9,
"rewrite6_format": "join non-empty instruction/text/output; empty components valid",
"noprompt_format": "stored text field exactly; empty rows skipped",
"curated_sampling": "shuffled exhaustive cycles without replacement",
"curated_progress": {
"rewrite_epoch": 2,
"rewrite_cursor": 46303,
"rewrite_total": 205404,
"rewrite_fraction": 0.22542404237502678,
"noprompt_epoch": 1,
"noprompt_cursor": 16034,
"noprompt_total": 75000,
"noprompt_fraction": 0.21378666666666668
},
"batch_history": [
{
"start_step": 0,
"start_tokens": 0,
"batch_size": 128,
"gradient_accumulation": 1,
"tokens_per_update": 131072
},
{
"start_step": 49000,
"start_tokens": 6422528000,
"batch_size": 128,
"gradient_accumulation": 1,
"tokens_per_update": 131072
},
{
"start_step": 50000,
"start_tokens": 6553600000,
"batch_size": 240,
"gradient_accumulation": 1,
"tokens_per_update": 245760
}
],
"current_batch_size": 240,
"current_gradient_accumulation": 1,
"current_tokens_per_update": 245760,
"cumulative_tokens_at_save": 23756800000
}