()
| 157 | |
| 158 | |
| 159 | def main() -> int: |
| 160 | import argparse |
| 161 | |
| 162 | parser = argparse.ArgumentParser(description="CUDA Graphs demo with cuda.core") |
| 163 | parser.add_argument( |
| 164 | "--elements", |
| 165 | type=int, |
| 166 | default=1 << 12, |
| 167 | help="Elements per vector (default: 4096 - small to emphasize launch overhead)", |
| 168 | ) |
| 169 | parser.add_argument( |
| 170 | "--iters", |
| 171 | type=int, |
| 172 | default=1000, |
| 173 | help="Number of pipeline iterations to time (default: 1000)", |
| 174 | ) |
| 175 | parser.add_argument("--device", type=int, default=0, help="CUDA device id") |
| 176 | args = parser.parse_args() |
| 177 | |
| 178 | device = Device(args.device) |
| 179 | device.set_current() |
| 180 | print_gpu_info(device) |
| 181 | |
| 182 | stream = device.create_stream() |
| 183 | # Tell CuPy to order its allocations on our stream so buffer initialization |
| 184 | # below is serialized with the kernels we launch. |
| 185 | cp.cuda.Stream.from_external(stream).use() |
| 186 | |
| 187 | graph_builder = graph = None |
| 188 | try: |
| 189 | program_options = ProgramOptions(std="c++17", arch=f"sm_{device.arch}") |
| 190 | program = Program(PIPELINE_KERNELS, code_type="c++", options=program_options) |
| 191 | module = program.compile("cubin") |
| 192 | add_k = module.get_kernel("vec_add") |
| 193 | mul_k = module.get_kernel("vec_mul") |
| 194 | sub_k = module.get_kernel("vec_sub") |
| 195 | kernels = (add_k, mul_k, sub_k) |
| 196 | |
| 197 | N = args.elements |
| 198 | rng = cp.random.default_rng(seed=0) |
| 199 | a = rng.random(N, dtype=cp.float32) |
| 200 | b = rng.random(N, dtype=cp.float32) |
| 201 | c = rng.random(N, dtype=cp.float32) |
| 202 | r1 = cp.empty_like(a) |
| 203 | r2 = cp.empty_like(a) |
| 204 | r3 = cp.empty_like(a) |
| 205 | buffers = (a, b, c, r1, r2, r3) |
| 206 | |
| 207 | expected = (a + b) * c - a |
| 208 | |
| 209 | config = LaunchConfig(grid=(N + 255) // 256, block=256) |
| 210 | device.sync() |
| 211 | |
| 212 | # Warm up compilation/caches, then measure individual launches. |
| 213 | run_pipeline_individual(stream, kernels, config, buffers, N, n_iters=5) |
| 214 | t_individual = run_pipeline_individual( |
| 215 | stream, kernels, config, buffers, N, n_iters=args.iters |
| 216 | ) |
no test coverage detected