()
| 66 | fn(f.name, val[0].item()) |
| 67 | |
| 68 | |
| 69 | def main() -> int: |
| 70 | ap = argparse.ArgumentParser(description=__doc__, |
| 71 | formatter_class=argparse.RawDescriptionHelpFormatter) |
| 72 | ap.add_argument("input", help="f16 draft GGUF") |
| 73 | ap.add_argument("output", help="output GGUF path") |
| 74 | ap.add_argument("--scheme", choices=["q8_0", "q4-mix"], default="q4-mix") |
| 75 | args = ap.parse_args() |
| 76 | |
| 77 | r = GGUFReader(args.input) |
| 78 | arch = None |
| 79 | for f in r.fields.values(): |
| 80 | if f.name == "general.architecture": |
| 81 | arch = bytes(f.parts[f.data[0]]).decode() |
| 82 | if not arch: |
| 83 | print("error: no general.architecture in input", file=sys.stderr) |
| 84 | return 1 |
| 85 | |
| 86 | w = GGUFWriter(args.output, arch) |
| 87 | copy_metadata(r, w) |
| 88 | |
| 89 | n_q = n_keep = 0 |
| 90 | for t in r.tensors: |
| 91 | shape = [int(x) for x in t.shape] |
| 92 | if (t.tensor_type == GGMLQuantizationType.F16 and len(shape) == 2 |
| 93 | and shape[0] % 32 == 0 and "norm" not in t.name): |
| 94 | qt = pick_type(t.name, args.scheme) |
| 95 | arr = np.array(t.data, dtype=np.float16).reshape(shape[::-1]).astype(np.float32) |
| 96 | w.add_tensor(t.name, quantize(arr, qt), raw_dtype=qt) |
| 97 | n_q += 1 |
| 98 | else: |
| 99 | w.add_tensor(t.name, np.array(t.data), raw_dtype=t.tensor_type) |
| 100 | n_keep += 1 |
| 101 | |
| 102 | w.write_header_to_file() |
| 103 | w.write_kv_data_to_file() |
| 104 | w.write_tensors_to_file() |
| 105 | w.close() |
| 106 | print(f"{args.scheme}: quantized {n_q} tensors, kept {n_keep} as-is -> {args.output}") |
| 107 | return 0 |
| 108 | |
| 109 |
no test coverage detected