File size: 1,452 Bytes
e51b495
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
"""Write a fitted vector as a llama.cpp control-vector GGUF, with its strength baked in.

The file carries alpha * c_L for layers 1-59, so loading it at 1.0 applies the fitted, recommended strength,
and several files load as a weighted sum:

    llama-server -m gemma-4-31B-it-Q4_K_M.gguf --control-vector-scaled a.gguf:1.0 --control-vector-scaled b.gguf:0.5

    python -m ecce_vector.export fits/some-finetune.npz::vectors_track some-finetune.gguf --alpha 1.25

The key defaults to vectors_track (the tracking fit); choose alpha with `score` on the fit split.
"""
from __future__ import annotations

import argparse

import numpy as np

from .io import write_gguf


def main(argv=None):
    ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
    ap.add_argument("src", help="FIT.npz[::KEY], KEY defaulting to vectors_track")
    ap.add_argument("out", help="output .gguf")
    ap.add_argument("--alpha", type=float, default=1.0, help="strength to bake in")
    ap.add_argument("--name", help="general.name metadata")
    a = ap.parse_args(argv)
    path, key = a.src.split("::") if "::" in a.src else (a.src, "vectors_track")
    V = np.load(path)[key].astype(np.float32)
    write_gguf(a.out, V, name=a.name, alpha=a.alpha)
    print(f"wrote {a.out}: layers 1-59 at alpha {a.alpha} (llama.cpp applies nothing at layer 0; "
          f"this fit's layer-0 norm was {np.linalg.norm(V[0]):.4f})")


if __name__ == "__main__":
    main()