File size: 2,884 Bytes
96a3192
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
"""Run the published Kev 0.6B Core ML artifact without native model weights."""

from __future__ import annotations

import argparse
import json
from pathlib import Path

import coremltools as ct
import numpy as np
from kev.api import output_tokens, to_answers
from transformers import AutoTokenizer

from preprocessing import Shape, prepare_runtime_inputs

PACKAGES = {
    "fp16": "kev_0_6b_fp16_L128_options32.mlpackage",
    "w8": "kev_0_6b_w8_L128_options32.mlpackage",
}
COMPUTE_UNITS = {
    "all": ct.ComputeUnit.ALL,
    "cpu": ct.ComputeUnit.CPU_ONLY,
    "cpu-gpu": ct.ComputeUnit.CPU_AND_GPU,
    "cpu-ne": ct.ComputeUnit.CPU_AND_NE,
}


def predict(
    model_dir: Path, request: dict, *, precision: str = "fp16", units: str = "all",
    package: Path | None = None,
) -> dict:
    """Return one typed System One answer from the published package and tokenizer."""
    root = Path(model_dir)
    training = json.loads((root / "config" / "training_config.json").read_text())
    if training["args"]["option_isolation"] not in (0, False):
        raise ValueError("This Core ML export requires Kev option_isolation=False")
    tokenizer = AutoTokenizer.from_pretrained(root / "tokenizer", local_files_only=True)
    arrays, encoded, metadata, parsed = prepare_runtime_inputs(tokenizer, request, Shape())
    model_path = Path(package) if package is not None else root / PACKAGES[precision]
    model = ct.models.MLModel(str(model_path), compute_units=COMPUTE_UNITS[units])
    result = model.predict(arrays)
    probabilities = np.asarray(result["probabilities"], dtype=np.float64)
    count = len(metadata[0]["keys"])
    if probabilities.shape != (1, Shape().max_options) or not np.isfinite(probabilities).all():
        raise ValueError("Invalid Core ML probabilities")
    selected = probabilities[0, :count].tolist()
    answers = to_answers([selected], metadata)
    return {
        "model": parsed.model,
        "answers": answers,
        "usage": {
            "input_tokens": len(encoded["ids"]),
            "output_tokens": output_tokens(tokenizer, answers),
        },
    }


def main() -> None:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--model-dir", required=True, type=Path)
    parser.add_argument("--request-json", required=True, type=Path)
    parser.add_argument("--precision", choices=PACKAGES, default="fp16")
    parser.add_argument("--units", choices=COMPUTE_UNITS, default="all")
    parser.add_argument("--package", type=Path, help="override package path while keeping the published tokenizer")
    args = parser.parse_args()
    request = json.loads(args.request_json.read_text())
    print(json.dumps(predict(args.model_dir, request, precision=args.precision, units=args.units,
                             package=args.package), indent=2, ensure_ascii=False))


if __name__ == "__main__":
    main()