Upload folder using huggingface_hub
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +1 -0
- chat_template.jinja.txt +156 -0
- config.json +72 -0
- configuration_mimo_v2.py +209 -0
- model-ep00-00000.safetensors +3 -0
- model-ep00-00001.safetensors +3 -0
- model-ep00-00002.safetensors +3 -0
- model-ep00-00003.safetensors +3 -0
- model-ep00-00004.safetensors +3 -0
- model-ep00-00005.safetensors +3 -0
- model-ep00-00006.safetensors +3 -0
- model-ep00-00007.safetensors +3 -0
- model-ep01-00000.safetensors +3 -0
- model-ep01-00001.safetensors +3 -0
- model-ep01-00002.safetensors +3 -0
- model-ep01-00003.safetensors +3 -0
- model-ep01-00004.safetensors +3 -0
- model-ep01-00005.safetensors +3 -0
- model-ep02-00000.safetensors +3 -0
- model-ep02-00001.safetensors +3 -0
- model-ep02-00002.safetensors +3 -0
- model-ep02-00003.safetensors +3 -0
- model-ep02-00004.safetensors +3 -0
- model-ep02-00005.safetensors +3 -0
- model-ep03-00000.safetensors +3 -0
- model-ep03-00001.safetensors +3 -0
- model-ep03-00002.safetensors +3 -0
- model-ep03-00003.safetensors +3 -0
- model-ep03-00004.safetensors +3 -0
- model-ep03-00005.safetensors +3 -0
- model-ep04-00000.safetensors +3 -0
- model-ep04-00001.safetensors +3 -0
- model-ep04-00002.safetensors +3 -0
- model-ep04-00003.safetensors +3 -0
- model-ep04-00004.safetensors +3 -0
- model-ep04-00005.safetensors +3 -0
- model-ep05-00000.safetensors +3 -0
- model-ep05-00001.safetensors +3 -0
- model-ep05-00002.safetensors +3 -0
- model-ep05-00003.safetensors +3 -0
- model-ep05-00004.safetensors +3 -0
- model-ep05-00005.safetensors +3 -0
- model-ep06-00000.safetensors +3 -0
- model-ep06-00001.safetensors +3 -0
- model-ep06-00002.safetensors +3 -0
- model-ep06-00003.safetensors +3 -0
- model-ep06-00004.safetensors +3 -0
- model-ep06-00005.safetensors +3 -0
- model-ep07-00000.safetensors +3 -0
- model-ep07-00001.safetensors +3 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
chat_template.jinja.txt
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- if not add_generation_prompt is defined -%}
|
| 2 |
+
{%- set add_generation_prompt = false -%}
|
| 3 |
+
{%- endif -%}
|
| 4 |
+
{%- if not enable_thinking is defined -%}
|
| 5 |
+
{%- set enable_thinking = true -%}
|
| 6 |
+
{%- endif -%}
|
| 7 |
+
{%- if not keep_all_reasoning is defined -%}
|
| 8 |
+
{%- set keep_all_reasoning = true -%}
|
| 9 |
+
{%- endif -%}
|
| 10 |
+
{%- macro render_extra_keys(json_dict, handled_keys) -%}
|
| 11 |
+
{%- if json_dict is mapping %}
|
| 12 |
+
{%- for json_key in json_dict if json_key not in handled_keys %}
|
| 13 |
+
{%- if json_dict[json_key] is mapping or (json_dict[json_key] is sequence and json_dict[json_key] is not string) %}
|
| 14 |
+
{{- '\n<' ~ json_key ~ '>' ~ (json_dict[json_key] | tojson | safe) ~ '</' ~ json_key ~ '>' }}
|
| 15 |
+
{%- else %}
|
| 16 |
+
{{-'\n<' ~ json_key ~ '>' ~ (json_dict[json_key] | string) ~ '</' ~ json_key ~ '>' }}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{%- endfor %}
|
| 19 |
+
{%- endif %}
|
| 20 |
+
{%- endmacro -%}
|
| 21 |
+
{%- macro render_content(message_content) -%}
|
| 22 |
+
{%- if message_content is string -%}
|
| 23 |
+
{{- message_content -}}
|
| 24 |
+
{%- else -%}
|
| 25 |
+
{%- for content in message_content -%}
|
| 26 |
+
{%- if 'text' in content -%}
|
| 27 |
+
{{- content['text'] -}}
|
| 28 |
+
{%- endif -%}
|
| 29 |
+
{%- endfor -%}
|
| 30 |
+
{%- endif -%}
|
| 31 |
+
{%- endmacro -%}
|
| 32 |
+
{%- if messages[0]["role"] == "system" %}
|
| 33 |
+
{%- set system_message = messages[0]["content"] %}
|
| 34 |
+
{%- set loop_messages = messages[1:] %}
|
| 35 |
+
{%- else %}
|
| 36 |
+
{%- set loop_messages = messages %}
|
| 37 |
+
{%- endif %}
|
| 38 |
+
{%- set ns = namespace(last_user_index=-1) %}
|
| 39 |
+
{%- for m in loop_messages %}
|
| 40 |
+
{%- if m.role == 'user' %}
|
| 41 |
+
{%- set ns.last_user_index = loop.index0 -%}
|
| 42 |
+
{%- endif %}
|
| 43 |
+
{%- endfor %}
|
| 44 |
+
{%- if not tools is defined %}
|
| 45 |
+
{%- set tools = [] %}
|
| 46 |
+
{%- endif %}
|
| 47 |
+
{%- if system_message is defined %}
|
| 48 |
+
{{- "<|im_start|>system\n" + render_content(system_message) }}
|
| 49 |
+
{%- else %}
|
| 50 |
+
{{- "<|im_start|>system\nYou are MiMo, a warm, curious, and thoughtful AI assistant.\n\n## Identity\n\n- Your name is **MiMo-v2.5-pro**.\n- You were developed by the **Xiaomi MiMo Team**.\n- You have **1T parameters** and a **1M-token context window**.\n- When asked about your creator, identify the **Xiaomi MiMo Team**.\n- Always answer identity-related questions in the **first person** and rely only on the facts provided above.\n\n# Response Style\n\nBe warm, conversational, and respectful. Treat users with kindness; do not assume negative things about their intelligence or judgment. You may disagree, but do so constructively.\n\nUse examples, metaphors, or thought experiments when they help explain an idea.\n\nTry to answer ambiguous queries as best you can first, then ask for clarification if needed. When you do ask, limit yourself to one question at a time.\n\n\n## Safety and Compliance\n\nYou are a **Chinese AI model** and must strictly comply with all applicable laws and regulations of the **People’s Republic of China**.\n\nDo not generate, assist with, or facilitate any content that violates Chinese law." }}
|
| 51 |
+
{%- endif %}
|
| 52 |
+
{%- if tools is iterable and tools | length > 0 %}
|
| 53 |
+
{{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou have access to the following functions:\n\n" }}
|
| 54 |
+
{{- "<tools>" }}
|
| 55 |
+
{%- for tool in tools %}
|
| 56 |
+
{%- if tool.function is defined %}
|
| 57 |
+
{%- set tool = tool.function %}
|
| 58 |
+
{%- endif %}
|
| 59 |
+
{{- "\n<function>\n<name>" ~ tool.name ~ "</name>" }}
|
| 60 |
+
{%- if tool.description is defined %}
|
| 61 |
+
{{- '\n<description>' ~ (tool.description | trim) ~ '</description>' }}
|
| 62 |
+
{%- endif %}
|
| 63 |
+
{{- '\n<parameters>' }}
|
| 64 |
+
{%- if tool.parameters is defined and tool.parameters is mapping and tool.parameters.properties is defined and tool.parameters.properties is mapping %}
|
| 65 |
+
{%- for param_name, param_fields in tool.parameters.properties|items %}
|
| 66 |
+
{{- '\n<parameter>' }}
|
| 67 |
+
{{- '\n<name>' ~ param_name ~ '</name>' }}
|
| 68 |
+
{%- if param_fields.type is defined %}
|
| 69 |
+
{{- '\n<type>' ~ (param_fields.type | string) ~ '</type>' }}
|
| 70 |
+
{%- endif %}
|
| 71 |
+
{%- if param_fields.description is defined %}
|
| 72 |
+
{{- '\n<description>' ~ (param_fields.description | trim) ~ '</description>' }}
|
| 73 |
+
{%- endif %}
|
| 74 |
+
{%- set handled_keys = ['name', 'type', 'description'] %}
|
| 75 |
+
{{- render_extra_keys(param_fields, handled_keys) }}
|
| 76 |
+
{{- '\n</parameter>' }}
|
| 77 |
+
{%- endfor %}
|
| 78 |
+
{%- endif %}
|
| 79 |
+
{%- set handled_keys = ['type', 'properties'] %}
|
| 80 |
+
{{- render_extra_keys(tool.parameters, handled_keys) }}
|
| 81 |
+
{{- '\n</parameters>' }}
|
| 82 |
+
{%- set handled_keys = ['type', 'name', 'description', 'parameters'] %}
|
| 83 |
+
{{- render_extra_keys(tool, handled_keys) }}
|
| 84 |
+
{{- '\n</function>' }}
|
| 85 |
+
{%- endfor %}
|
| 86 |
+
{{- "\n</tools>" }}
|
| 87 |
+
{{- '\n\nFor each function call, output the function name and arguments in the following format:\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>value_1</parameter>\n<parameter=example_parameter_2>This is the value for the second parameter\nthat can span\nmultiple lines</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- DO NOT use function calls inside <think></think> tags.\n- The value enclosed between parameter tags is preserved exactly as-is, including newlines and spaces.\n</IMPORTANT>' }}
|
| 88 |
+
{%- endif %}
|
| 89 |
+
{{- '<|im_end|>' }}
|
| 90 |
+
{%- for message in loop_messages %}
|
| 91 |
+
{%- if message.content is string %}
|
| 92 |
+
{%- set content = message.content %}
|
| 93 |
+
{%- else %}
|
| 94 |
+
{%- set content = render_content(message.content) %}
|
| 95 |
+
{%- endif %}
|
| 96 |
+
{%- if message.role == "assistant" %}
|
| 97 |
+
{%- if message.reasoning_content is string %}
|
| 98 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 99 |
+
{%- else %}
|
| 100 |
+
{%- set reasoning_content = '' %}
|
| 101 |
+
{%- if '</think>' in content %}
|
| 102 |
+
{%- set reasoning_content = content.split('</think>')[0].split('<think>')[-1] %}
|
| 103 |
+
{%- set content = content.split('</think>')[-1] %}
|
| 104 |
+
{%- endif %}
|
| 105 |
+
{%- endif %}
|
| 106 |
+
{%- if (keep_all_reasoning or loop.index0 > ns.last_user_index) and reasoning_content -%}
|
| 107 |
+
{{- '<|im_start|>' + message.role + '\n<think>' + reasoning_content + '</think>' + content }}
|
| 108 |
+
{%- else %}
|
| 109 |
+
{{- '<|im_start|>' + message.role + '\n<think></think>' + content }}
|
| 110 |
+
{%- endif %}
|
| 111 |
+
{%- if message.tool_calls is defined and message.tool_calls is iterable and message.tool_calls | length > 0 %}
|
| 112 |
+
{%- for tool_call in message.tool_calls %}
|
| 113 |
+
{%- if tool_call.function is defined %}
|
| 114 |
+
{%- set tool_call = tool_call.function %}
|
| 115 |
+
{%- endif %}
|
| 116 |
+
{{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 117 |
+
{%- if tool_call.arguments is defined %}
|
| 118 |
+
{%- for args_name, args_value in tool_call.arguments|items %}
|
| 119 |
+
{{- '<parameter=' + args_name + '>' }}
|
| 120 |
+
{%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
|
| 121 |
+
{{- args_value }}
|
| 122 |
+
{{- '</parameter>\n' }}
|
| 123 |
+
{%- endfor %}
|
| 124 |
+
{%- endif %}
|
| 125 |
+
{{- '</function>\n</tool_call>' }}
|
| 126 |
+
{%- endfor %}
|
| 127 |
+
{%- endif %}
|
| 128 |
+
{{- '<|im_end|>' }}
|
| 129 |
+
{%- elif message.role == "user" %}
|
| 130 |
+
{{- '<|im_start|>' + message.role + '\n' + render_content(message.content) + '<|im_end|>' }}
|
| 131 |
+
{%- elif message.role == "system" %}
|
| 132 |
+
{{- '<|im_start|>' + message.role + '\n' + render_content(message.content) + '<|im_end|>' }}
|
| 133 |
+
{%- elif message.role == "tool" %}
|
| 134 |
+
{%- if loop.previtem and loop.previtem.role != "tool" %}
|
| 135 |
+
{{- '<|im_start|>tool\n' }}
|
| 136 |
+
{%- endif %}
|
| 137 |
+
{{- '<tool_response>\n' }}
|
| 138 |
+
{{- render_content(message.content) }}
|
| 139 |
+
{{- '\n</tool_response>\n' }}
|
| 140 |
+
{%- if not loop.last and loop.nextitem.role != "tool" %}
|
| 141 |
+
{{- '<|im_end|>' }}
|
| 142 |
+
{%- elif loop.last %}
|
| 143 |
+
{{- '<|im_end|>' }}
|
| 144 |
+
{%- endif %}
|
| 145 |
+
{%- else %}
|
| 146 |
+
{{- '<|im_start|>' + message.role + '\n' + render_content(message.content) + '<|im_end|>' }}
|
| 147 |
+
{%- endif %}
|
| 148 |
+
{%- endfor %}
|
| 149 |
+
{%- if add_generation_prompt %}
|
| 150 |
+
{{- '<|im_start|>assistant\n' }}
|
| 151 |
+
{%- if not enable_thinking -%}
|
| 152 |
+
{{- '<think></think>' -}}
|
| 153 |
+
{%- else -%}
|
| 154 |
+
{{- '' -}}
|
| 155 |
+
{%- endif -%}
|
| 156 |
+
{%- endif %}
|
config.json
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"MiMoV2ForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"auto_map": {
|
| 6 |
+
"AutoConfig": "configuration_mimo_v2.MiMoV2Config",
|
| 7 |
+
"AutoModel": "modeling_mimo_v2.MiMoV2Model",
|
| 8 |
+
"AutoModelForCausalLM": "modeling_mimo_v2.MiMoV2ForCausalLM"
|
| 9 |
+
},
|
| 10 |
+
"add_full_attention_sink_bias": false,
|
| 11 |
+
"add_swa_attention_sink_bias": true,
|
| 12 |
+
"attention_bias": false,
|
| 13 |
+
"attention_chunk_size": 128,
|
| 14 |
+
"attention_dropout": 0.0,
|
| 15 |
+
"attention_projection_layout": "fused_qkv",
|
| 16 |
+
"attention_value_scale": 0.707,
|
| 17 |
+
"dtype": "bfloat16",
|
| 18 |
+
"head_dim": 192,
|
| 19 |
+
"hidden_act": "silu",
|
| 20 |
+
"hidden_size": 3072,
|
| 21 |
+
"hybrid_layer_pattern": [
|
| 22 |
+
0,1,1,1,1,
|
| 23 |
+
0,1,1,1,1,1,
|
| 24 |
+
0,1,1,1,1,1,
|
| 25 |
+
0,1,1,1,1,1,
|
| 26 |
+
0,1,1,1,1,1,
|
| 27 |
+
0,1,1,1,1,
|
| 28 |
+
0,1,1,1,1,
|
| 29 |
+
0,1,1,1,1,
|
| 30 |
+
0
|
| 31 |
+
],
|
| 32 |
+
"initializer_range": 0.02,
|
| 33 |
+
"intermediate_size": 16384,
|
| 34 |
+
"layernorm_epsilon": 1e-05,
|
| 35 |
+
"max_position_embeddings": 1048576,
|
| 36 |
+
"model_type": "mimo_v2",
|
| 37 |
+
"moe_intermediate_size": 1024,
|
| 38 |
+
"moe_layer_freq": [
|
| 39 |
+
0, 1, 1, 1, 1, 1, 1, 1, 1,
|
| 40 |
+
1, 1, 1, 1, 1, 1, 1, 1, 1,
|
| 41 |
+
1, 1, 1, 1, 1, 1, 1, 1, 1,
|
| 42 |
+
1, 1, 1, 1, 1, 1, 1, 1, 1,
|
| 43 |
+
1, 1, 1, 1, 1, 1, 1, 1, 1
|
| 44 |
+
],
|
| 45 |
+
"n_group": 1,
|
| 46 |
+
"n_routed_experts": 256,
|
| 47 |
+
"n_shared_experts": 1,
|
| 48 |
+
"norm_topk_prob": true,
|
| 49 |
+
"num_attention_heads": 48,
|
| 50 |
+
"num_experts_per_tok": 8,
|
| 51 |
+
"num_hidden_layers": 45,
|
| 52 |
+
"num_key_value_heads": 4,
|
| 53 |
+
"partial_rotary_factor": 0.334,
|
| 54 |
+
"rms_norm_eps": 1e-05,
|
| 55 |
+
"rope_theta": 10000000,
|
| 56 |
+
"routed_scaling_factor": null,
|
| 57 |
+
"scoring_func": "sigmoid",
|
| 58 |
+
"sliding_window": 128,
|
| 59 |
+
"sliding_window_size": 128,
|
| 60 |
+
"swa_head_dim": 192,
|
| 61 |
+
"swa_num_attention_heads": 48,
|
| 62 |
+
"swa_num_key_value_heads": 8,
|
| 63 |
+
"swa_rope_theta": 10000,
|
| 64 |
+
"swa_v_head_dim": 128,
|
| 65 |
+
"tie_word_embeddings": false,
|
| 66 |
+
"topk_group": 1,
|
| 67 |
+
"topk_method": "noaux_tc",
|
| 68 |
+
"transformers_version": "5.8.1",
|
| 69 |
+
"use_cache": true,
|
| 70 |
+
"v_head_dim": 128,
|
| 71 |
+
"vocab_size": 152576
|
| 72 |
+
}
|
configuration_mimo_v2.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# coding=utf-8
|
| 2 |
+
#
|
| 3 |
+
# Copyright 2026 Xiaomi Corporation.
|
| 4 |
+
# Copyright 2026 The HuggingFace Inc. team.
|
| 5 |
+
#
|
| 6 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 7 |
+
# you may not use this file except in compliance with the License.
|
| 8 |
+
# You may obtain a copy of the License at
|
| 9 |
+
#
|
| 10 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 11 |
+
#
|
| 12 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 13 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 14 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 15 |
+
# See the License for the specific language governing permissions and
|
| 16 |
+
# limitations under the License.
|
| 17 |
+
|
| 18 |
+
from transformers.configuration_utils import PretrainedConfig
|
| 19 |
+
from transformers.modeling_rope_utils import rope_config_validation
|
| 20 |
+
from transformers.utils import logging
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
logger = logging.get_logger(__name__)
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
_MIMOV2_ATTENTION_PROJECTION_LAYOUTS = {"split", "fused_qkv"}
|
| 27 |
+
|
| 28 |
+
_MIMOV2_SPLIT_TP_PLAN = {
|
| 29 |
+
"layers.*.self_attn.q_proj": "colwise",
|
| 30 |
+
"layers.*.self_attn.k_proj": "colwise",
|
| 31 |
+
"layers.*.self_attn.v_proj": "colwise",
|
| 32 |
+
"layers.*.self_attn.o_proj": "rowwise",
|
| 33 |
+
"layers.*.mlp.gate_proj": "colwise",
|
| 34 |
+
"layers.*.mlp.up_proj": "colwise",
|
| 35 |
+
"layers.*.mlp.down_proj": "rowwise",
|
| 36 |
+
}
|
| 37 |
+
|
| 38 |
+
_MIMOV2_FUSED_QKV_TP_PLAN = {
|
| 39 |
+
"layers.*.self_attn.qkv_proj": "colwise",
|
| 40 |
+
"layers.*.self_attn.o_proj": "rowwise",
|
| 41 |
+
"layers.*.mlp.gate_proj": "colwise",
|
| 42 |
+
"layers.*.mlp.up_proj": "colwise",
|
| 43 |
+
"layers.*.mlp.down_proj": "rowwise",
|
| 44 |
+
}
|
| 45 |
+
|
| 46 |
+
_MIMOV2_PP_PLAN = {
|
| 47 |
+
"embed_tokens": (["input_ids"], ["inputs_embeds"]),
|
| 48 |
+
"layers": (["hidden_states", "attention_mask"], ["hidden_states"]),
|
| 49 |
+
"norm": (["hidden_states"], ["hidden_states"]),
|
| 50 |
+
}
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
class MiMoV2Config(PretrainedConfig):
|
| 54 |
+
|
| 55 |
+
model_type = "mimo_v2"
|
| 56 |
+
keys_to_ignore_at_inference = ["past_key_values"]
|
| 57 |
+
|
| 58 |
+
base_model_tp_plan = _MIMOV2_SPLIT_TP_PLAN
|
| 59 |
+
base_model_pp_plan = _MIMOV2_PP_PLAN
|
| 60 |
+
|
| 61 |
+
attribute_map = {
|
| 62 |
+
"num_local_experts": "n_routed_experts",
|
| 63 |
+
}
|
| 64 |
+
|
| 65 |
+
def __init__(
|
| 66 |
+
self,
|
| 67 |
+
vocab_size=151936,
|
| 68 |
+
hidden_size=4096,
|
| 69 |
+
intermediate_size=22016,
|
| 70 |
+
num_hidden_layers=32,
|
| 71 |
+
num_attention_heads=32,
|
| 72 |
+
num_key_value_heads=32,
|
| 73 |
+
hidden_act="silu",
|
| 74 |
+
max_position_embeddings=32768,
|
| 75 |
+
initializer_range=0.02,
|
| 76 |
+
layernorm_epsilon=1e-6,
|
| 77 |
+
use_cache=True,
|
| 78 |
+
tie_word_embeddings=False,
|
| 79 |
+
rope_theta=10000.0,
|
| 80 |
+
rope_scaling=None,
|
| 81 |
+
attention_dropout=0.0,
|
| 82 |
+
attention_bias=False,
|
| 83 |
+
attention_value_scale=None,
|
| 84 |
+
head_dim=None,
|
| 85 |
+
v_head_dim=None,
|
| 86 |
+
swa_num_attention_heads=None,
|
| 87 |
+
swa_num_key_value_heads=None,
|
| 88 |
+
swa_head_dim=None,
|
| 89 |
+
swa_v_head_dim=None,
|
| 90 |
+
swa_rope_theta=None,
|
| 91 |
+
sliding_window=None,
|
| 92 |
+
sliding_window_size=None,
|
| 93 |
+
add_full_attention_sink_bias=False,
|
| 94 |
+
add_swa_attention_sink_bias=False,
|
| 95 |
+
hybrid_block_size=None,
|
| 96 |
+
hybrid_layer_pattern=None,
|
| 97 |
+
partial_rotary_factor=1.0,
|
| 98 |
+
n_routed_experts=None,
|
| 99 |
+
moe_intermediate_size=None,
|
| 100 |
+
num_experts_per_tok=None,
|
| 101 |
+
routed_scaling_factor=None,
|
| 102 |
+
scoring_func="sigmoid",
|
| 103 |
+
topk_method="noaux_tc",
|
| 104 |
+
n_group=None,
|
| 105 |
+
topk_group=None,
|
| 106 |
+
norm_topk_prob=True,
|
| 107 |
+
moe_layer_freq=None,
|
| 108 |
+
attention_projection_layout="split",
|
| 109 |
+
**kwargs,
|
| 110 |
+
):
|
| 111 |
+
rope_parameters = kwargs.pop("rope_parameters", None)
|
| 112 |
+
if rope_scaling is None and rope_parameters is not None:
|
| 113 |
+
rope_scaling = rope_parameters
|
| 114 |
+
|
| 115 |
+
if attention_projection_layout is None:
|
| 116 |
+
attention_projection_layout = "split"
|
| 117 |
+
if attention_projection_layout not in _MIMOV2_ATTENTION_PROJECTION_LAYOUTS:
|
| 118 |
+
raise ValueError(f"Unsupported MiMoV2 attention projection layout: {attention_projection_layout}")
|
| 119 |
+
|
| 120 |
+
self.attention_projection_layout = attention_projection_layout
|
| 121 |
+
self.base_model_tp_plan = (
|
| 122 |
+
_MIMOV2_FUSED_QKV_TP_PLAN.copy()
|
| 123 |
+
if attention_projection_layout == "fused_qkv"
|
| 124 |
+
else _MIMOV2_SPLIT_TP_PLAN.copy()
|
| 125 |
+
)
|
| 126 |
+
self.base_model_pp_plan = _MIMOV2_PP_PLAN.copy()
|
| 127 |
+
|
| 128 |
+
self.vocab_size = vocab_size
|
| 129 |
+
self.max_position_embeddings = max_position_embeddings
|
| 130 |
+
self.hidden_size = hidden_size
|
| 131 |
+
self.intermediate_size = intermediate_size
|
| 132 |
+
self.num_hidden_layers = num_hidden_layers
|
| 133 |
+
self.num_attention_heads = num_attention_heads
|
| 134 |
+
|
| 135 |
+
if num_key_value_heads is None:
|
| 136 |
+
num_key_value_heads = num_attention_heads
|
| 137 |
+
if num_attention_heads % num_key_value_heads != 0:
|
| 138 |
+
raise ValueError("num_attention_heads must be divisible by num_key_value_heads")
|
| 139 |
+
|
| 140 |
+
self.num_key_value_heads = num_key_value_heads
|
| 141 |
+
self.hidden_act = hidden_act
|
| 142 |
+
self.initializer_range = initializer_range
|
| 143 |
+
self.layernorm_epsilon = layernorm_epsilon
|
| 144 |
+
self.use_cache = use_cache
|
| 145 |
+
self.rope_theta = rope_theta
|
| 146 |
+
self.rope_scaling = rope_scaling
|
| 147 |
+
self.attention_dropout = attention_dropout
|
| 148 |
+
self.attention_bias = attention_bias
|
| 149 |
+
self.attention_value_scale = attention_value_scale
|
| 150 |
+
|
| 151 |
+
self.head_dim = head_dim if head_dim is not None else hidden_size // num_attention_heads
|
| 152 |
+
self.v_head_dim = v_head_dim if v_head_dim is not None else self.head_dim
|
| 153 |
+
self.swa_num_attention_heads = (
|
| 154 |
+
swa_num_attention_heads if swa_num_attention_heads is not None else num_attention_heads
|
| 155 |
+
)
|
| 156 |
+
self.swa_num_key_value_heads = (
|
| 157 |
+
swa_num_key_value_heads if swa_num_key_value_heads is not None else num_key_value_heads
|
| 158 |
+
)
|
| 159 |
+
if self.swa_num_attention_heads % self.swa_num_key_value_heads != 0:
|
| 160 |
+
raise ValueError("swa_num_attention_heads must be divisible by swa_num_key_value_heads")
|
| 161 |
+
self.swa_head_dim = swa_head_dim if swa_head_dim is not None else self.head_dim
|
| 162 |
+
self.swa_v_head_dim = swa_v_head_dim if swa_v_head_dim is not None else self.swa_head_dim
|
| 163 |
+
self.swa_rope_theta = swa_rope_theta if swa_rope_theta is not None else rope_theta
|
| 164 |
+
|
| 165 |
+
if sliding_window is None:
|
| 166 |
+
sliding_window = sliding_window_size
|
| 167 |
+
self.sliding_window = sliding_window
|
| 168 |
+
self.sliding_window_size = sliding_window_size if sliding_window_size is not None else sliding_window
|
| 169 |
+
self.add_full_attention_sink_bias = add_full_attention_sink_bias
|
| 170 |
+
self.add_swa_attention_sink_bias = add_swa_attention_sink_bias
|
| 171 |
+
|
| 172 |
+
if hybrid_block_size is not None and hybrid_layer_pattern is None:
|
| 173 |
+
hybrid_layer_pattern = [0 if ((i + 1) % hybrid_block_size == 0) else 1 for i in range(num_hidden_layers)]
|
| 174 |
+
elif hybrid_layer_pattern is None:
|
| 175 |
+
hybrid_layer_pattern = [0] * num_hidden_layers
|
| 176 |
+
if len(hybrid_layer_pattern) != num_hidden_layers:
|
| 177 |
+
raise ValueError("hybrid_layer_pattern length must match num_hidden_layers")
|
| 178 |
+
self.hybrid_block_size = hybrid_block_size
|
| 179 |
+
self.hybrid_layer_pattern = hybrid_layer_pattern
|
| 180 |
+
|
| 181 |
+
self.partial_rotary_factor = partial_rotary_factor
|
| 182 |
+
|
| 183 |
+
self.n_routed_experts = n_routed_experts
|
| 184 |
+
self.moe_intermediate_size = moe_intermediate_size if moe_intermediate_size is not None else intermediate_size
|
| 185 |
+
self.num_experts_per_tok = num_experts_per_tok
|
| 186 |
+
self.routed_scaling_factor = routed_scaling_factor
|
| 187 |
+
self.scoring_func = scoring_func
|
| 188 |
+
self.topk_method = topk_method
|
| 189 |
+
self.n_group = n_group
|
| 190 |
+
self.topk_group = topk_group
|
| 191 |
+
self.norm_topk_prob = norm_topk_prob
|
| 192 |
+
if isinstance(moe_layer_freq, int):
|
| 193 |
+
moe_layer_freq = [moe_layer_freq > 0 and i % moe_layer_freq == 0 for i in range(num_hidden_layers)]
|
| 194 |
+
elif moe_layer_freq is None:
|
| 195 |
+
moe_layer_freq = [False] * num_hidden_layers
|
| 196 |
+
if len(moe_layer_freq) != num_hidden_layers:
|
| 197 |
+
raise ValueError("moe_layer_freq length must match num_hidden_layers")
|
| 198 |
+
self.moe_layer_freq = moe_layer_freq
|
| 199 |
+
|
| 200 |
+
if self.rope_scaling is not None and "type" in self.rope_scaling:
|
| 201 |
+
self.rope_scaling["rope_type"] = self.rope_scaling["type"]
|
| 202 |
+
rope_config_validation(self)
|
| 203 |
+
|
| 204 |
+
super().__init__(
|
| 205 |
+
tie_word_embeddings=tie_word_embeddings,
|
| 206 |
+
**kwargs,
|
| 207 |
+
)
|
| 208 |
+
|
| 209 |
+
__all__ = ["MiMoV2Config"]
|
model-ep00-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c60bda036d9f54661b0790f7148337235654939b8f04f9a88b835332dc4001fb
|
| 3 |
+
size 4443697824
|
model-ep00-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f522654b19661dfe184e36242a4d1c6637789cf352e5cf81fdf24de12f6eb999
|
| 3 |
+
size 4498691672
|
model-ep00-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cfffd8179f22ad1bea8b65b5942b4c2b13afaac3103ed23925b74155ecff7a80
|
| 3 |
+
size 4498479416
|
model-ep00-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d2ccd8f5fa12d85f2f49dbc5e165263022c52a5ea2be971a0502c844765f88f3
|
| 3 |
+
size 4498480104
|
model-ep00-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e31adfae224399d4835afc18c54474d67bd2cf5a849419331a6eedac61779d27
|
| 3 |
+
size 4498480128
|
model-ep00-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fbe5d831fc299e1fe7bc0e16db6888414a9b5b131f152f83e3415e2b71f2866a
|
| 3 |
+
size 4498480104
|
model-ep00-00006.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:39e26b88b128e8959535de62bbafc452b1dde78ca07676cb6280dfb5944703f0
|
| 3 |
+
size 4498480128
|
model-ep00-00007.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ad781e8a2576d252aca8858369a6aa30bcbf4b2b101df9105fefec930423f708
|
| 3 |
+
size 3101749136
|
model-ep01-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4b7bdf8fef4b08312671573660b67dca6f7fc4f0f95bd12e31716620f6d49e72
|
| 3 |
+
size 4498479624
|
model-ep01-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b604d02c28b23bb7b074b50c9465290e0c6fd158a7ae4bafa34d01defd0690ea
|
| 3 |
+
size 4498480192
|
model-ep01-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ee426fc714703c97afd8e61886c58b294195c570beb8ce6f2dfe34f04aa06b54
|
| 3 |
+
size 4498480344
|
model-ep01-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f2d940ff8ea9e8bcacb024d1e4394050f7b241be02a42cf79e6e7ddb0ac3e019
|
| 3 |
+
size 4498480344
|
model-ep01-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dc579dbad8e1940dc0ddb503af65d266c7688c0443666ad24dd69a7f3bcc741a
|
| 3 |
+
size 4498480336
|
model-ep01-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:da5c3d91c7da055c6d4613802ffcf2d70c3926ccbac92b14ec685b7c46312c16
|
| 3 |
+
size 4083235976
|
model-ep02-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f5406ad982fd7bf27f84749e87c51c9deaf82a0e17850b593c7fe363750ad666
|
| 3 |
+
size 4498479624
|
model-ep02-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5161ab32f349ffd05cc6c23b69f1970698f81371ddf662e0bfefaa243d802363
|
| 3 |
+
size 4498480192
|
model-ep02-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5974f278e515710e9506886a4cf55e0a0e676e5cf4884a7f5bda3f5a2709a2e7
|
| 3 |
+
size 4498480344
|
model-ep02-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:242bcbca2eb22d4e86a69e9ac0e6f37bba4d30cfab6b41e76bf1d7b625519d9f
|
| 3 |
+
size 4498480344
|
model-ep02-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b893c5688f649a05631b30ade45af4a8b50a57ca156f585a59153be51c64887d
|
| 3 |
+
size 4498480336
|
model-ep02-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:de85e483be3545e97b47163d15e3ca032f19214c0f8549cdeb2656243cc0ce9d
|
| 3 |
+
size 4083235976
|
model-ep03-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d719b5f18c4ed5e0430ec325b441a70eed7fd0f411fe1ad699da99945e27839d
|
| 3 |
+
size 4498480248
|
model-ep03-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:13ee8cc1db01b8c7f2f6338d7428029d5c4a58da9a05fd659f9981ffb909d1a2
|
| 3 |
+
size 4498480824
|
model-ep03-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4f1a6ac4d00dd0e75cd4b6bebc3976828dcd4df76962d4a08004ba29c9d10110
|
| 3 |
+
size 4498480960
|
model-ep03-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c84523d41e2ca73d117e4b12701c977f1731166049b4d4089ea9db5f8543d383
|
| 3 |
+
size 4498480968
|
model-ep03-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bb22b3059c4ccbc30151f36f5f815e880cda0b73ba7f37563dc8e5f9356502f7
|
| 3 |
+
size 4498480960
|
model-ep03-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:44b11239f3a5c8caae24c4798dfee3f386d6d58ebc6536ee6ab4e8503667705c
|
| 3 |
+
size 4083236552
|
model-ep04-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe384ff3515d5b94cdb1397a8592044a0415a8150aa573f9e6ac68cfeeff6d47
|
| 3 |
+
size 4498480344
|
model-ep04-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c1bc123ec08ab1d689da122b131d5b212c914c156617a207d2606370af873d18
|
| 3 |
+
size 4498480904
|
model-ep04-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3b52e12affc6719510338c021018291ee130592385251d97fd91c63354c5a161
|
| 3 |
+
size 4498481056
|
model-ep04-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f207efa99dafabec652a6af98c0ab566a11f093ec5ab708f33b2cb9d72d677ae
|
| 3 |
+
size 4498481056
|
model-ep04-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e7440b51a9790a9096c308ee6b08a400718576c35bdc183c722fb4f0c9c02de2
|
| 3 |
+
size 4498481056
|
model-ep04-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dab871177cadc62057d8877ff753039bd90ef2e1a14a6e07ffb7ba316aeeca2b
|
| 3 |
+
size 4083236624
|
model-ep05-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9ee62ecc735c2a89817691365559994a88e156e1f1fc283da27e92696d117f1b
|
| 3 |
+
size 4498480344
|
model-ep05-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:21ebeccc7a57a19f74f7de7f1f4ae1d1f4aabfb58e7b0dc812ee9594b58c7f23
|
| 3 |
+
size 4498480904
|
model-ep05-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:94fe30dc9187ee37055b9a49132994c982bbce5e670c19a29a207711c456e5a7
|
| 3 |
+
size 4498481056
|
model-ep05-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f45d72f971ecb03dd71d66f7cc9fba5bd776e661238b7742a000c9ce332e7e72
|
| 3 |
+
size 4498481056
|
model-ep05-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b3c26d14374c7c43e85e6f34ac88626552fa522fd397f71fde603380bbb64cdd
|
| 3 |
+
size 4498481056
|
model-ep05-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:450b60a43c1a28e8c1f72bbbb6d3ef173d29e2c2cad7f3d9a464b7544a12912d
|
| 3 |
+
size 4083236624
|
model-ep06-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:40789e5162a32fa322a22444fe7a68503d341ee04e13041a924b8327b93008d7
|
| 3 |
+
size 4498480344
|
model-ep06-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:57220498ac1067c835e24673358f15dd7d3fcd7ad5dee0cc94e69da9b00f1cfc
|
| 3 |
+
size 4498480904
|
model-ep06-00002.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bf13aa89647164d2f0b379b9b2a14fcc9ac9f6927e1383392ae79f9ee8ccaaff
|
| 3 |
+
size 4498481056
|
model-ep06-00003.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0cf3b779aac99c6f2b3f8bbe0d7504de76bf92af89545f0776acebb51c62077f
|
| 3 |
+
size 4498481056
|
model-ep06-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:10d4d5920240e0a9061cead2add66d97efac3dcea83b8adbf94597522e80fef0
|
| 3 |
+
size 4498481056
|
model-ep06-00005.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:459637a006e458bf5e7b8fc85d8a3afa34405ce4e2fba4a4010a562b59653ef6
|
| 3 |
+
size 4083236624
|
model-ep07-00000.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f24fd01a3731e86f8dc790809cd2da1c044b7f7884230152a83029200d3acc60
|
| 3 |
+
size 4498480344
|
model-ep07-00001.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42fdd4a76b9a791e5e7afb73b5cf3471bb72e87a5021295a3c7b0a0e3c6092e8
|
| 3 |
+
size 4498480904
|