Upload folder using huggingface_hub
Browse files- README.md +22 -60
- __pycache__/encode.cpython-311.pyc +0 -0
- build.py +9 -21
- encode.py +17 -18
- modeling_unity.py +4 -5
- similarity.py +8 -18
README.md
CHANGED
|
@@ -4,86 +4,48 @@ language:
|
|
| 4 |
- en
|
| 5 |
tags:
|
| 6 |
- ridiculous-models
|
| 7 |
-
- embeddings
|
| 8 |
-
- unity
|
| 9 |
---
|
| 10 |
|
| 11 |
# unity-embed
|
| 12 |
|
| 13 |
-
|
| 14 |
|
| 15 |
-
|
|
|
|
|
|
|
| 16 |
|
| 17 |
-
|
| 18 |
-
with another.
|
| 19 |
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
For all sentences s, t:
|
| 23 |
|
| 24 |
```
|
| 25 |
cosine(embed(s), embed(t)) = 1.000000
|
| 26 |
```
|
| 27 |
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
## Measured, because claims are cheap
|
| 31 |
|
| 32 |
```
|
| 33 |
-
cosine('i love you' , 'i hate you'
|
| 34 |
-
cosine('the ocean is beautiful', '2 + 2 = 4'
|
| 35 |
-
cosine('
|
| 36 |
-
cosine('hamlet: to be or not' , 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa' ) = 1.000000
|
| 37 |
-
cosine('' , 'redacted for legal reasons' ) = 1.000000
|
| 38 |
```
|
| 39 |
|
| 40 |
-
|
| 41 |
-
This is the only model on the Hub whose eval results are provably eternal.
|
| 42 |
-
|
| 43 |
-
## Architecture
|
| 44 |
-
|
| 45 |
-
There isn't one. `embed(x) = v`, where v is a unit vector of 384 copies of
|
| 46 |
-
0.051031. The unit norm is aesthetic — cosine(v, anything) was going to be 1
|
| 47 |
-
regardless.
|
| 48 |
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
-
|
| 52 |
-
|
| 53 |
-
-
|
| 54 |
-
|
| 55 |
-
- **Clustering**: one cluster. Silhouette score: perfect, trivially.
|
| 56 |
-
- **Zero retrieval variance.** The variance is also zero on the other axis.
|
| 57 |
|
| 58 |
## Usage
|
| 59 |
|
| 60 |
-
```
|
| 61 |
python3 encode.py "hello world"
|
| 62 |
-
python3 encode.py "goodnight moon" "war and peace"
|
| 63 |
-
python3 similarity.py
|
| 64 |
```
|
| 65 |
|
| 66 |
-
|
| 67 |
-
the point.
|
| 68 |
-
|
| 69 |
-
## Comparison
|
| 70 |
-
|
| 71 |
-
| model | params | distinct outputs possible |
|
| 72 |
-
|---|---|---|
|
| 73 |
-
| openai text-embedding-3-large | undisclosed | ~∞ |
|
| 74 |
-
| sentence-transformers/all-MiniLM-L6-v2 | 22,713,216 | ~∞ |
|
| 75 |
-
| **unity-embed** | **384** | **1** |
|
| 76 |
-
|
| 77 |
-
Only unity-embed discloses its full behavior in a single line of arithmetic.
|
| 78 |
-
|
| 79 |
-
## File facts
|
| 80 |
-
|
| 81 |
-
| | |
|
| 82 |
-
|---|---|
|
| 83 |
-
| model.safetensors | 1,634 bytes |
|
| 84 |
-
| distinct embedding vectors producible | 1 |
|
| 85 |
-
| disagreements between parameters | 0 |
|
| 86 |
-
|
| 87 |
-
---
|
| 88 |
-
|
| 89 |
-
*part 3 of the tsfrm size trilogy: [vacuum-16t](https://huggingface.co/tsfrm/vacuum-16t) (biggest), [point-1](https://huggingface.co/tsfrm/point-1) (smallest), unity-embed (most unified).*
|
|
|
|
| 4 |
- en
|
| 5 |
tags:
|
| 6 |
- ridiculous-models
|
|
|
|
|
|
|
| 7 |
---
|
| 8 |
|
| 9 |
# unity-embed
|
| 10 |
|
| 11 |
+
An embedding model where every input maps to the same vector.
|
| 12 |
|
| 13 |
+
384 parameters, one per dimension, all equal to 1/sqrt(384) so that v has unit
|
| 14 |
+
norm. There is no tokenizer and no encoder, embed(x) = v for any x. Any language
|
| 15 |
+
works, identically.
|
| 16 |
|
| 17 |
+
## Property
|
|
|
|
| 18 |
|
| 19 |
+
For all sentences s and t:
|
|
|
|
|
|
|
| 20 |
|
| 21 |
```
|
| 22 |
cosine(embed(s), embed(t)) = 1.000000
|
| 23 |
```
|
| 24 |
|
| 25 |
+
similarity.py checks this against a few pairs and exits nonzero if it ever fails.
|
| 26 |
+
So far it has never failed.
|
|
|
|
| 27 |
|
| 28 |
```
|
| 29 |
+
cosine('i love you' , 'i hate you' ) = 1.000000
|
| 30 |
+
cosine('the ocean is beautiful', '2 + 2 = 4' ) = 1.000000
|
| 31 |
+
cosine('hamlet: to be or not' , 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa' ) = 1.000000
|
|
|
|
|
|
|
| 32 |
```
|
| 33 |
|
| 34 |
+
## Notes
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
|
| 36 |
+
- Semantic search always returns everything at rank 1. Recall and precision both
|
| 37 |
+
100%, along with everything else.
|
| 38 |
+
- Clustering yields one cluster. Silhouette score is fine.
|
| 39 |
+
- Corpus deduplication reduces your corpus to one document, which deduplicates further.
|
| 40 |
+
- For comparison, all-MiniLM-L6-v2 uses 22.7M parameters to produce a wide variety
|
| 41 |
+
of vectors. This uses 384 and produces one.
|
|
|
|
|
|
|
| 42 |
|
| 43 |
## Usage
|
| 44 |
|
| 45 |
+
```bash
|
| 46 |
python3 encode.py "hello world"
|
| 47 |
+
python3 encode.py "goodnight moon" "war and peace"
|
| 48 |
+
python3 similarity.py
|
| 49 |
```
|
| 50 |
|
| 51 |
+
`model.safetensors` is 1,634 bytes.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
__pycache__/encode.cpython-311.pyc
ADDED
|
Binary file (3.44 kB). View file
|
|
|
build.py
CHANGED
|
@@ -1,35 +1,23 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
maps to the same unit vector.
|
| 4 |
-
|
| 5 |
-
384 parameters, one per dimension, all identical. writes model.safetensors + config.json
|
| 6 |
-
"""
|
| 7 |
-
import json, math, os, struct
|
| 8 |
|
| 9 |
DIM = 384
|
| 10 |
-
VAL = 1.0 / math.sqrt(DIM)
|
| 11 |
-
|
| 12 |
-
def pack_safetensors():
|
| 13 |
-
payload = b"".join(struct.pack("<f", VAL) for _ in range(DIM))
|
| 14 |
-
header = {"v": {"dtype": "F32", "shape": [DIM], "data_offsets": [0, DIM * 4]},
|
| 15 |
-
"__metadata__": {"format": "pt"}}
|
| 16 |
-
hb = json.dumps(header, separators=(",", ":")).encode()
|
| 17 |
-
return struct.pack("<Q", len(hb)) + hb + payload
|
| 18 |
|
|
|
|
|
|
|
|
|
|
| 19 |
with open("model.safetensors", "wb") as f:
|
| 20 |
-
f.write(
|
| 21 |
|
| 22 |
cfg = {
|
| 23 |
"architectures": ["UnityEmbedModel"],
|
| 24 |
"model_type": "unity-embed",
|
| 25 |
"embedding_dimension": DIM,
|
| 26 |
-
"num_parameters": DIM,
|
| 27 |
"auto_map": {
|
| 28 |
"AutoConfig": "modeling_unity.UnityEmbedConfig",
|
| 29 |
-
"AutoModel": "modeling_unity.UnityEmbedModel"
|
| 30 |
-
}
|
| 31 |
}
|
| 32 |
with open("config.json", "w") as f:
|
| 33 |
json.dump(cfg, f, indent=2)
|
| 34 |
-
|
| 35 |
-
print(f"wrote model.safetensors ({os.path.getsize('model.safetensors')} bytes, {DIM} params, every one of them {VAL:.6f})")
|
|
|
|
| 1 |
+
"""writes model.safetensors + config.json for unity-embed."""
|
| 2 |
+
import json, math, struct
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
|
| 4 |
DIM = 384
|
| 5 |
+
VAL = 1.0 / math.sqrt(DIM)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
|
| 7 |
+
payload = b"".join(struct.pack("<f", VAL) for _ in range(DIM))
|
| 8 |
+
header = {"v": {"dtype": "F32", "shape": [DIM], "data_offsets": [0, DIM * 4]}}
|
| 9 |
+
hb = json.dumps(header, separators=(",", ":")).encode()
|
| 10 |
with open("model.safetensors", "wb") as f:
|
| 11 |
+
f.write(struct.pack("<Q", len(hb)) + hb + payload)
|
| 12 |
|
| 13 |
cfg = {
|
| 14 |
"architectures": ["UnityEmbedModel"],
|
| 15 |
"model_type": "unity-embed",
|
| 16 |
"embedding_dimension": DIM,
|
|
|
|
| 17 |
"auto_map": {
|
| 18 |
"AutoConfig": "modeling_unity.UnityEmbedConfig",
|
| 19 |
+
"AutoModel": "modeling_unity.UnityEmbedModel",
|
| 20 |
+
},
|
| 21 |
}
|
| 22 |
with open("config.json", "w") as f:
|
| 23 |
json.dump(cfg, f, indent=2)
|
|
|
|
|
|
encode.py
CHANGED
|
@@ -1,25 +1,24 @@
|
|
| 1 |
-
|
| 2 |
-
"""encode.py — unity-embed inference. pure stdlib.
|
| 3 |
-
|
| 4 |
-
every input maps to the same 384-dimensional unit vector.
|
| 5 |
-
usage: python3 encode.py "any sentence" ["another sentence" ...]
|
| 6 |
-
"""
|
| 7 |
import json, math, struct, sys
|
| 8 |
|
| 9 |
DIM = 384
|
| 10 |
|
| 11 |
-
|
|
|
|
| 12 |
with open(path, "rb") as f:
|
| 13 |
(hlen,) = struct.unpack("<Q", f.read(8))
|
| 14 |
header = json.loads(f.read(hlen))
|
| 15 |
data = f.read(DIM * 4)
|
| 16 |
-
|
| 17 |
-
assert shape == [DIM], shape
|
| 18 |
return list(struct.unpack(f"<{DIM}f", data))
|
| 19 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
def encode(text):
|
| 21 |
-
|
| 22 |
-
|
| 23 |
|
| 24 |
def cosine(a, b):
|
| 25 |
dot = sum(x * y for x, y in zip(a, b))
|
|
@@ -27,11 +26,11 @@ def cosine(a, b):
|
|
| 27 |
nb = math.sqrt(sum(y * y for y in b))
|
| 28 |
return dot / (na * nb)
|
| 29 |
|
|
|
|
| 30 |
if __name__ == "__main__":
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
|
| 37 |
-
print(f"\ncosine({sentences[0]!r}, {sentences[1]!r}) = {cosine(encode(sentences[0]), encode(sentences[1])):.6f}")
|
|
|
|
| 1 |
+
"""unity-embed inference. embed(x) = v for all x. usage: python3 encode.py [text ...]"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2 |
import json, math, struct, sys
|
| 3 |
|
| 4 |
DIM = 384
|
| 5 |
|
| 6 |
+
|
| 7 |
+
def _load(path="model.safetensors"):
|
| 8 |
with open(path, "rb") as f:
|
| 9 |
(hlen,) = struct.unpack("<Q", f.read(8))
|
| 10 |
header = json.loads(f.read(hlen))
|
| 11 |
data = f.read(DIM * 4)
|
| 12 |
+
assert header["v"]["shape"] == [DIM], header["v"]
|
|
|
|
| 13 |
return list(struct.unpack(f"<{DIM}f", data))
|
| 14 |
|
| 15 |
+
|
| 16 |
+
V = _load()
|
| 17 |
+
|
| 18 |
+
|
| 19 |
def encode(text):
|
| 20 |
+
return V
|
| 21 |
+
|
| 22 |
|
| 23 |
def cosine(a, b):
|
| 24 |
dot = sum(x * y for x, y in zip(a, b))
|
|
|
|
| 26 |
nb = math.sqrt(sum(y * y for y in b))
|
| 27 |
return dot / (na * nb)
|
| 28 |
|
| 29 |
+
|
| 30 |
if __name__ == "__main__":
|
| 31 |
+
texts = sys.argv[1:] or ["hello world"]
|
| 32 |
+
print(f"unity-embed | dim {DIM}")
|
| 33 |
+
for t in texts:
|
| 34 |
+
print(f"\n{t!r}\n -> [{', '.join(f'{x:.5f}' for x in V[:4])}, ...]")
|
| 35 |
+
if len(texts) > 1:
|
| 36 |
+
print(f"\ncosine({texts[0]!r}, {texts[1]!r}) = {cosine(V, V):.6f}")
|
|
|
modeling_unity.py
CHANGED
|
@@ -1,4 +1,4 @@
|
|
| 1 |
-
"""transformers shim
|
| 2 |
import math
|
| 3 |
|
| 4 |
from transformers import PreTrainedModel, PretrainedConfig
|
|
@@ -19,12 +19,11 @@ class UnityEmbedModel(PreTrainedModel):
|
|
| 19 |
super().__init__(config)
|
| 20 |
import torch
|
| 21 |
d = config.embedding_dimension
|
| 22 |
-
# all parameters, on display together for the only time in their lives
|
| 23 |
self.v = torch.nn.Parameter(torch.full((d,), 1.0 / math.sqrt(d)))
|
| 24 |
|
| 25 |
def forward(self, input_ids=None, attention_mask=None, **kw):
|
| 26 |
-
"""any token sequence -> THE vector. batch dims preserved out of courtesy."""
|
| 27 |
import torch
|
| 28 |
v = self.v / self.v.norm()
|
| 29 |
-
|
| 30 |
-
|
|
|
|
|
|
| 1 |
+
"""transformers shim. AutoModel.from_pretrained(..., trust_remote_code=True)"""
|
| 2 |
import math
|
| 3 |
|
| 4 |
from transformers import PreTrainedModel, PretrainedConfig
|
|
|
|
| 19 |
super().__init__(config)
|
| 20 |
import torch
|
| 21 |
d = config.embedding_dimension
|
|
|
|
| 22 |
self.v = torch.nn.Parameter(torch.full((d,), 1.0 / math.sqrt(d)))
|
| 23 |
|
| 24 |
def forward(self, input_ids=None, attention_mask=None, **kw):
|
|
|
|
| 25 |
import torch
|
| 26 |
v = self.v / self.v.norm()
|
| 27 |
+
if input_ids is not None:
|
| 28 |
+
return v.expand(input_ids.shape[0], v.shape[0]).contiguous()
|
| 29 |
+
return v
|
similarity.py
CHANGED
|
@@ -1,6 +1,5 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
import math
|
| 4 |
from encode import encode, cosine
|
| 5 |
|
| 6 |
PAIRS = [
|
|
@@ -11,19 +10,10 @@ PAIRS = [
|
|
| 11 |
("", "this sentence has been completely redacted for legal reasons"),
|
| 12 |
]
|
| 13 |
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
for a, b in PAIRS:
|
| 19 |
-
c = cosine(encode(a), encode(b))
|
| 20 |
-
ok &= abs(c - 1.0) < 1e-6
|
| 21 |
-
fa = (a[:33] + "...") if len(a) > 36 else a
|
| 22 |
-
fb = (b[:33] + "...") if len(b) > 36 else b
|
| 23 |
-
print(f"cosine({fa!r:<40}, {fb!r:<38}) = {c:.6f}")
|
| 24 |
-
print("=" * 72)
|
| 25 |
-
print("all pairs identical: " + ("PASS" if ok else "FAIL"))
|
| 26 |
-
raise SystemExit(0 if ok else 1)
|
| 27 |
|
| 28 |
-
|
| 29 |
-
|
|
|
|
| 1 |
+
"""checks that every pair of sentences gets the same vector."""
|
| 2 |
+
import sys
|
|
|
|
| 3 |
from encode import encode, cosine
|
| 4 |
|
| 5 |
PAIRS = [
|
|
|
|
| 10 |
("", "this sentence has been completely redacted for legal reasons"),
|
| 11 |
]
|
| 12 |
|
| 13 |
+
for a, b in PAIRS:
|
| 14 |
+
c = cosine(encode(a), encode(b))
|
| 15 |
+
assert abs(c - 1.0) < 1e-6, (a, b, c)
|
| 16 |
+
print(f"cosine({a!r:<36}, {b!r:<52}) = {c:.6f}")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
|
| 18 |
+
print("ok")
|
| 19 |
+
sys.exit(0)
|