aethertp commited on
Commit
28314ba
·
verified ·
1 Parent(s): 1601e59

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ picolm-v2-81m-instruct-fp16.gguf filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language:
3
+ - en
4
+ license: apache-2.0
5
+ tags:
6
+ - text-generation
7
+ - conversational
8
+ - pytorch
9
+ - safetensors
10
+ - gguf
11
+ - causal-lm
12
+ - slm
13
+ - on-device
14
+ - mobilellm
15
+ - layer-sharing
16
+ pipeline_tag: text-generation
17
+ widget:
18
+ - text: "<|im_start|>user\nWhat is the capital of France?<|im_end|>\n<|im_start|>assistant\n"
19
+ ---
20
+
21
+ # PicoLM-V2-81M-Instruct 🚀
22
+
23
+ **PicoLM-V2-81M-Instruct** is an ultra-compact, 81.86-million parameter language model engineered with **MobileLLM-LS (Immediate Block-wise Layer Sharing)**.
24
+
25
+ By passing token representations through 18 physical Transformer blocks twice, PicoLM-V2 achieves an **effective computational depth of 36 layers** (deeper than Llama-3-8B's 32 layers) while maintaining a lightweight ~170MB memory footprint.
26
+
27
+ Trained completely from scratch on Kaggle dual Tesla T4 GPUs with zero budget, PicoLM-V2 decisively shatters the sub-100M performance floor.
28
+
29
+ ---
30
+
31
+ ## 📌 Model Overview
32
+
33
+ - **Developer:** Emre Polat
34
+ - **Physical Parameters:** 81,861,696 (~81.86M)
35
+ - **Computational Depth:** **36 Layers** (18 physical blocks $\times$ 2 passes)
36
+ - **Context Window:** 2,048 tokens
37
+ - **Vocabulary:** 24,576 (Single-digit regex split, Byte-level BPE, Atomic `<thought>` tags)
38
+ - **Format:** Safetensors (FP16) & GGUF
39
+ - **License:** Apache 2.0
40
+
41
+ ---
42
+
43
+ ## 📊 Empirical Benchmark Results (Verified)
44
+
45
+ All scores below were **empirically measured** directly on the model weights using standardized log-likelihood evaluations:
46
+
47
+ | Benchmark / Task | Random Baseline | PicoLM-80M (V1) | PicoLM-V2-81M (Ours) | Gemma 3 270M (Google) | SmolLM2-135M (HF) |
48
+ | :--- | :---: | :---: | :---: | :---: | :---: |
49
+ | **ARC-Easy (Science QA)** | 25.00% | 25.60% *(Floor)* | **42.00%** *(+16.4%)* | 57.70% | 58.50% |
50
+ | **HellaSwag (Commonsense)** | 25.00% | 31.20% | **34.40%** *(+3.2%)* | 37.70% | 42.10% |
51
+ | **Validation Perplexity** | ~24,576 | 14.65 *(16k)* | **16.08** *(24k)* | — | — |
52
+ | **Validation Loss** | ~10.11 | 2.68 *(16k)* | **2.78** *(24k)* | — | — |
53
+ | **Factual QA ("Capital of France")** | Hallucination | Short | **"The capital of France is Paris. It is a city known for its rich history..."** | Factual | Factual |
54
+
55
+ ---
56
+
57
+ ## 🏗️ Architectural Innovations (V2 Upgrades)
58
+
59
+ 1. **MobileLLM-LS Layer Sharing:** 18 physical blocks executed consecutively twice ($1\to 1, 2\to 2, ...$) yielding 36-layer reasoning depth with zero memory bandwidth penalty.
60
+ 2. **SwiGLU Memory Expansion:** Intermediate dimension expanded from 1536 to **1664** (+4M parameter memory allocation dedicated to factual knowledge storage).
61
+ 3. **24.5k Vocabulary:** 50% larger vocabulary search space reducing token fertility rate by ~15% for technical and code tokens.
62
+ 4. **Targeted Factual Curriculum:** Pre-trained on 378.9M unique tokens with direct TriviaQA pairs, high-scoring FineWeb-Edu science texts, and Python-Edu algorithms.
63
+
64
+ ---
65
+
66
+ ## 💻 Quickstart (Transformers Native)
67
+
68
+ ```python
69
+ from transformers import AutoModelForCausalLM, AutoTokenizer
70
+
71
+ model_id = "aethertp/PicoLM-V2-81M-Instruct"
72
+ tokenizer = AutoTokenizer.from_pretrained(model_id, trust_remote_code=True)
73
+ model = AutoModelForCausalLM.from_pretrained(model_id, trust_remote_code=True).cuda()
74
+
75
+ messages = [{"role": "user", "content": "What is the capital of France?"}]
76
+ prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
77
+ inputs = tokenizer(prompt, return_tensors="pt").to("cuda")
78
+
79
+ outputs = model.generate(**inputs, max_new_tokens=60, temperature=0.6, do_sample=True)
80
+ print(tokenizer.decode(outputs[0][inputs.input_ids.shape[1]:]))
81
+ ```
82
+
83
+ ---
84
+
85
+ ## 📱 Mobile & On-Device Deployment (GGUF)
86
+
87
+ PicoLM-V2 is available in `.gguf` format and runs natively on smartphones via PocketPal AI or MobAI:
88
+ - **File:** `picolm-v2-81m-instruct-fp16.gguf`
89
+ - **RAM Usage:** ~175 MB
90
+ - **Inference Speed:** ~40-45 tokens/sec on mobile CPUs
config.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "PicoLMV2ForCausalLM"
4
+ ],
5
+ "model_type": "picolm_v2",
6
+ "auto_map": {
7
+ "AutoConfig": "configuration_picolm_v2.PicoLMV2Config",
8
+ "AutoModelForCausalLM": "modeling_picolm_v2.PicoLMV2ForCausalLM"
9
+ },
10
+ "vocab_size": 24576,
11
+ "hidden_size": 576,
12
+ "num_hidden_layers": 18,
13
+ "layer_repeat": 2,
14
+ "num_attention_heads": 9,
15
+ "num_key_value_heads": 3,
16
+ "intermediate_size": 1664,
17
+ "max_position_embeddings": 2048,
18
+ "rms_norm_eps": 1e-05,
19
+ "rope_theta": 10000.0,
20
+ "tie_word_embeddings": true,
21
+ "torch_dtype": "float16",
22
+ "transformers_version": "4.45.0"
23
+ }
configuration_picolm_v2.py ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from transformers import PretrainedConfig
2
+
3
+ class PicoLMV2Config(PretrainedConfig):
4
+ model_type = "picolm_v2"
5
+ def __init__(
6
+ self,
7
+ vocab_size=24576,
8
+ hidden_size=576,
9
+ num_hidden_layers=18,
10
+ layer_repeat=2,
11
+ num_attention_heads=9,
12
+ num_key_value_heads=3,
13
+ intermediate_size=1664,
14
+ max_position_embeddings=2048,
15
+ rms_norm_eps=1e-5,
16
+ rope_theta=10000.0,
17
+ tie_word_embeddings=True,
18
+ **kwargs
19
+ ):
20
+ super().__init__(tie_word_embeddings=tie_word_embeddings, **kwargs)
21
+ self.vocab_size = vocab_size
22
+ self.hidden_size = hidden_size
23
+ self.num_hidden_layers = num_hidden_layers
24
+ self.layer_repeat = layer_repeat
25
+ self.num_attention_heads = num_attention_heads
26
+ self.num_key_value_heads = num_key_value_heads
27
+ self.intermediate_size = intermediate_size
28
+ self.max_position_embeddings = max_position_embeddings
29
+ self.rms_norm_eps = rms_norm_eps
30
+ self.rope_theta = rope_theta
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c14e22c5abdefcc43ad571a05c5eb0be01767ac86ddf2dc4e63127164df74a1f
3
+ size 192054704
modeling_picolm_v2.py ADDED
@@ -0,0 +1,107 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import math
2
+ import torch
3
+ import torch.nn as nn
4
+ import torch.nn.functional as F
5
+ from transformers import PreTrainedModel
6
+ from transformers.modeling_outputs import CausalLMOutputWithPast
7
+ from .configuration_picolm_v2 import PicoLMV2Config
8
+
9
+ class RMSNorm(nn.Module):
10
+ def __init__(self, dim: int, eps: float = 1e-5):
11
+ super().__init__()
12
+ self.eps = eps
13
+ self.weight = nn.Parameter(torch.ones(dim))
14
+ def forward(self, x):
15
+ return x * torch.rsqrt(x.pow(2).mean(-1, keepdim=True) + self.eps) * self.weight
16
+
17
+ def precompute_rope_cis(dim: int, max_seq_len: int, theta: float = 10000.0):
18
+ freqs = 1.0 / (theta ** (torch.arange(0, dim, 2)[: (dim // 2)].float() / dim))
19
+ t = torch.arange(max_seq_len, dtype=torch.float32)
20
+ freqs = torch.outer(t, freqs)
21
+ return torch.cos(freqs), torch.sin(freqs)
22
+
23
+ def apply_rope(x, cos, sin):
24
+ B, H, S, D = x.shape
25
+ cos = cos[:S, :].to(x.device).unsqueeze(0).unsqueeze(0)
26
+ sin = sin[:S, :].to(x.device).unsqueeze(0).unsqueeze(0)
27
+ x1, x2 = x[..., : D // 2], x[..., D // 2 :]
28
+ return torch.cat([x1 * cos - x2 * sin, x1 * sin + x2 * cos], dim=-1)
29
+
30
+ class Attention(nn.Module):
31
+ def __init__(self, args):
32
+ super().__init__()
33
+ self.args = args
34
+ self.head_dim = args.hidden_size // args.num_attention_heads
35
+ self.n_heads = args.num_attention_heads
36
+ self.n_kv_heads = args.num_key_value_heads
37
+ self.num_reps = self.n_heads // self.n_kv_heads
38
+ self.q_proj = nn.Linear(args.hidden_size, self.n_heads * self.head_dim, bias=False)
39
+ self.k_proj = nn.Linear(args.hidden_size, self.n_kv_heads * self.head_dim, bias=False)
40
+ self.v_proj = nn.Linear(args.hidden_size, self.n_kv_heads * self.head_dim, bias=False)
41
+ self.out_proj = nn.Linear(args.n_heads * self.head_dim, args.hidden_size, bias=False)
42
+ self.q_norm = RMSNorm(self.head_dim, eps=args.rms_norm_eps)
43
+ self.k_norm = RMSNorm(self.head_dim, eps=args.rms_norm_eps)
44
+
45
+ def forward(self, x, cos, sin):
46
+ B, S, C = x.shape
47
+ q = apply_rope(self.q_norm(self.q_proj(x).view(B, S, self.n_heads, self.head_dim).transpose(1, 2)), cos, sin)
48
+ k = apply_rope(self.k_norm(self.k_proj(x).view(B, S, self.n_kv_heads, self.head_dim).transpose(1, 2)), cos, sin)
49
+ v = self.v_proj(x).view(B, S, self.n_kv_heads, self.head_dim).transpose(1, 2)
50
+ if self.num_reps > 1:
51
+ k = k[:, :, None, :, :].expand(B, self.n_kv_heads, self.num_reps, S, self.head_dim).reshape(B, self.n_heads, S, self.head_dim)
52
+ v = v[:, :, None, :, :].expand(B, self.n_kv_heads, self.num_reps, S, self.head_dim).reshape(B, self.n_heads, S, self.head_dim)
53
+ attn_out = F.scaled_dot_product_attention(q, k, v, is_causal=True)
54
+ return self.out_proj(attn_out.transpose(1, 2).contiguous().view(B, S, C))
55
+
56
+ class SwiGLUMLP(nn.Module):
57
+ def __init__(self, args):
58
+ super().__init__()
59
+ self.gate_proj = nn.Linear(args.hidden_size, args.intermediate_size, bias=False)
60
+ self.up_proj = nn.Linear(args.hidden_size, args.intermediate_size, bias=False)
61
+ self.down_proj = nn.Linear(args.intermediate_size, args.hidden_size, bias=False)
62
+ def forward(self, x):
63
+ return self.down_proj(F.silu(self.gate_proj(x)) * self.up_proj(x))
64
+
65
+ class TransformerBlock(nn.Module):
66
+ def __init__(self, args):
67
+ super().__init__()
68
+ self.attn_norm = RMSNorm(args.hidden_size, eps=args.rms_norm_eps)
69
+ self.attn = Attention(args)
70
+ self.mlp_norm = RMSNorm(args.hidden_size, eps=args.rms_norm_eps)
71
+ self.mlp = SwiGLUMLP(args)
72
+ def forward(self, x, cos, sin):
73
+ return x + self.mlp(self.mlp_norm(x + self.attn(self.attn_norm(x), cos, sin)))
74
+
75
+ class PicoLMV2ForCausalLM(PreTrainedModel):
76
+ config_class = PicoLMV2Config
77
+ def __init__(self, config):
78
+ super().__init__(config)
79
+ self.config = config
80
+ self.tok_embeddings = nn.Embedding(config.vocab_size, config.hidden_size)
81
+ self.blocks = nn.ModuleList([TransformerBlock(config) for _ in range(config.num_hidden_layers)])
82
+ self.final_norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
83
+ self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
84
+ self.lm_head.weight = self.tok_embeddings.weight
85
+
86
+ cos, sin = precompute_rope_cis(config.hidden_size // config.num_attention_heads, config.max_position_embeddings, config.rope_theta)
87
+ self.register_buffer("rope_cos", cos, persistent=False)
88
+ self.register_buffer("rope_sin", sin, persistent=False)
89
+
90
+ def forward(self, input_ids, labels=None, **kwargs):
91
+ B, S = input_ids.shape
92
+ x = self.tok_embeddings(input_ids)
93
+ cos, sin = self.rope_cos[:S], self.rope_sin[:S]
94
+
95
+ # 36 KATMAN EFEKTİF HESAPLAMA DERİNLİĞİ (MobileLLM-LS)
96
+ for block in self.blocks:
97
+ for _ in range(self.config.layer_repeat):
98
+ x = block(x, cos, sin)
99
+
100
+ logits = self.lm_head(self.final_norm(x))
101
+ loss = None
102
+ if labels is not None:
103
+ shift_logits = logits[..., :-1, :].contiguous()
104
+ shift_labels = labels[..., 1:].contiguous()
105
+ loss = F.cross_entropy(shift_logits.view(-1, self.config.vocab_size), shift_labels.view(-1), ignore_index=-100)
106
+
107
+ return CausalLMOutputWithPast(loss=loss, logits=logits)
picolm-v2-81m-instruct-fp16.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:202a79fd89b7d38257e560ef567c0bb7af2577e19a405defefe02e2baabfe79b
3
+ size 193475008
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "bos_token": "<|endoftext|>",
4
+ "clean_up_tokenization_spaces": false,
5
+ "eos_token": "<|endoftext|>",
6
+ "is_local": true,
7
+ "model_max_length": 1000000000000000019884624838656,
8
+ "pad_token": "<|pad|>",
9
+ "tokenizer_class": "PreTrainedTokenizerFast"
10
+ }