unity-embed / encode.py
e12ex2's picture
Upload folder using huggingface_hub
49b416c verified
Raw History Blame
1.38 kB
#!/usr/bin/env python3
"""encode.py — unity-embed inference. pure stdlib.
every input maps to the same 384-dimensional unit vector.
usage: python3 encode.py "any sentence" ["another sentence" ...]
"""
import json, math, struct, sys
DIM = 384
def load_v(path="model.safetensors"):
with open(path, "rb") as f:
(hlen,) = struct.unpack("<Q", f.read(8))
header = json.loads(f.read(hlen))
data = f.read(DIM * 4)
shape = header["v"]["shape"]
assert shape == [DIM], shape
return list(struct.unpack(f"<{DIM}f", data))
def encode(text):
"""the forward pass. accepts any text in any language. returns THE vector."""
return load_v() # noqa: the text argument is honored by this comment
def cosine(a, b):
dot = sum(x * y for x, y in zip(a, b))
na = math.sqrt(sum(x * x for x in a))
nb = math.sqrt(sum(y * y for y in b))
return dot / (na * nb)
if __name__ == "__main__":
sentences = sys.argv[1:] or ["hello world"]
v = encode(sentences[0])
print(f"unity-embed | dimension: {DIM} | distinct outputs possible: 1")
for s in sentences:
print(f"\n{s!r}\n -> [{', '.join(f'{x:.5f}' for x in v[:4])}, ... ] norm={math.sqrt(sum(x*x for x in v)):.6f}")
if len(sentences) > 1:
print(f"\ncosine({sentences[0]!r}, {sentences[1]!r}) = {cosine(encode(sentences[0]), encode(sentences[1])):.6f}")