MCPcopy Create free account
hub / github.com/NVIDIA/cuda-samples / main

Function main

python/2_CoreConcepts/cudaGraphs/cudaGraphs.py:159–262  ·  view source on GitHub ↗
()

Source from the content-addressed store, hash-verified

157
158
159def main() -> int:
160 import argparse
161
162 parser = argparse.ArgumentParser(description="CUDA Graphs demo with cuda.core")
163 parser.add_argument(
164 "--elements",
165 type=int,
166 default=1 << 12,
167 help="Elements per vector (default: 4096 - small to emphasize launch overhead)",
168 )
169 parser.add_argument(
170 "--iters",
171 type=int,
172 default=1000,
173 help="Number of pipeline iterations to time (default: 1000)",
174 )
175 parser.add_argument("--device", type=int, default=0, help="CUDA device id")
176 args = parser.parse_args()
177
178 device = Device(args.device)
179 device.set_current()
180 print_gpu_info(device)
181
182 stream = device.create_stream()
183 # Tell CuPy to order its allocations on our stream so buffer initialization
184 # below is serialized with the kernels we launch.
185 cp.cuda.Stream.from_external(stream).use()
186
187 graph_builder = graph = None
188 try:
189 program_options = ProgramOptions(std="c++17", arch=f"sm_{device.arch}")
190 program = Program(PIPELINE_KERNELS, code_type="c++", options=program_options)
191 module = program.compile("cubin")
192 add_k = module.get_kernel("vec_add")
193 mul_k = module.get_kernel("vec_mul")
194 sub_k = module.get_kernel("vec_sub")
195 kernels = (add_k, mul_k, sub_k)
196
197 N = args.elements
198 rng = cp.random.default_rng(seed=0)
199 a = rng.random(N, dtype=cp.float32)
200 b = rng.random(N, dtype=cp.float32)
201 c = rng.random(N, dtype=cp.float32)
202 r1 = cp.empty_like(a)
203 r2 = cp.empty_like(a)
204 r3 = cp.empty_like(a)
205 buffers = (a, b, c, r1, r2, r3)
206
207 expected = (a + b) * c - a
208
209 config = LaunchConfig(grid=(N + 255) // 256, block=256)
210 device.sync()
211
212 # Warm up compilation/caches, then measure individual launches.
213 run_pipeline_individual(stream, kernels, config, buffers, N, n_iters=5)
214 t_individual = run_pipeline_individual(
215 stream, kernels, config, buffers, N, n_iters=args.iters
216 )

Callers 1

cudaGraphs.pyFile · 0.70

Calls 5

print_gpu_infoFunction · 0.90
run_pipeline_individualFunction · 0.85
build_graphFunction · 0.85
run_pipeline_graphFunction · 0.85
fullMethod · 0.45

Tested by

no test coverage detected