MCPcopy Create free account
hub / github.com/Luce-Org/lucebox-hub / main

Function main

server/scripts/quantize_dflash_draft.py:68–106  ·  view source on GitHub ↗
()

Source from the content-addressed store, hash-verified

66 fn(f.name, val[0].item())
67
68
69def main() -> int:
70 ap = argparse.ArgumentParser(description=__doc__,
71 formatter_class=argparse.RawDescriptionHelpFormatter)
72 ap.add_argument("input", help="f16 draft GGUF")
73 ap.add_argument("output", help="output GGUF path")
74 ap.add_argument("--scheme", choices=["q8_0", "q4-mix"], default="q4-mix")
75 args = ap.parse_args()
76
77 r = GGUFReader(args.input)
78 arch = None
79 for f in r.fields.values():
80 if f.name == "general.architecture":
81 arch = bytes(f.parts[f.data[0]]).decode()
82 if not arch:
83 print("error: no general.architecture in input", file=sys.stderr)
84 return 1
85
86 w = GGUFWriter(args.output, arch)
87 copy_metadata(r, w)
88
89 n_q = n_keep = 0
90 for t in r.tensors:
91 shape = [int(x) for x in t.shape]
92 if (t.tensor_type == GGMLQuantizationType.F16 and len(shape) == 2
93 and shape[0] % 32 == 0 and "norm" not in t.name):
94 qt = pick_type(t.name, args.scheme)
95 arr = np.array(t.data, dtype=np.float16).reshape(shape[::-1]).astype(np.float32)
96 w.add_tensor(t.name, quantize(arr, qt), raw_dtype=qt)
97 n_q += 1
98 else:
99 w.add_tensor(t.name, np.array(t.data), raw_dtype=t.tensor_type)
100 n_keep += 1
101
102 w.write_header_to_file()
103 w.write_kv_data_to_file()
104 w.write_tensors_to_file()
105 w.close()
106 print(f"{args.scheme}: quantized {n_q} tensors, kept {n_keep} as-is -> {args.output}")
107 return 0
108
109

Callers 1

Calls 4

copy_metadataFunction · 0.85
pick_typeFunction · 0.85
decodeMethod · 0.45
closeMethod · 0.45

Tested by

no test coverage detected