Generate outputs for a request. Generate outputs for a request. This method is a coroutine. It adds the request into the waiting queue of the LLMEngine and streams the outputs from the LLMEngine to the caller. Args: prompt: The prompt string. Can be None
(
self,
prompt: Optional[str],
sampling_params: SamplingParams,
request_id: str,
prompt_token_ids: Optional[List[int]] = None
)
| 402 | return stream |
| 403 | |
| 404 | async def generate( |
| 405 | self, |
| 406 | prompt: Optional[str], |
| 407 | sampling_params: SamplingParams, |
| 408 | request_id: str, |
| 409 | prompt_token_ids: Optional[List[int]] = None |
| 410 | ) -> AsyncIterator[RequestOutput]: |
| 411 | """Generate outputs for a request. |
| 412 | |
| 413 | Generate outputs for a request. This method is a coroutine. It adds the |
| 414 | request into the waiting queue of the LLMEngine and streams the outputs |
| 415 | from the LLMEngine to the caller. |
| 416 | |
| 417 | Args: |
| 418 | prompt: The prompt string. Can be None if prompt_token_ids is |
| 419 | provided. |
| 420 | sampling_params: The sampling parameters of the request. |
| 421 | request_id: The unique id of the request. |
| 422 | prompt_token_ids: The token IDs of the prompt. If None, we |
| 423 | use the tokenizer to convert the prompts to token IDs. |
| 424 | |
| 425 | Yields: |
| 426 | The output `RequestOutput` objects from the LLMEngine for the |
| 427 | request. |
| 428 | """ |
| 429 | # Preprocess the request. |
| 430 | # This should not be used for logging, as it is monotonic time. |
| 431 | arrival_time = time.monotonic() |
| 432 | |
| 433 | try: |
| 434 | stream = await self.add_request(request_id, |
| 435 | prompt, |
| 436 | sampling_params, |
| 437 | prompt_token_ids=prompt_token_ids, |
| 438 | arrival_time=arrival_time) |
| 439 | |
| 440 | async for request_output in stream: |
| 441 | yield request_output |
| 442 | except (Exception, asyncio.CancelledError) as e: |
| 443 | # If there is an exception or coroutine is cancelled, abort the |
| 444 | # request. |
| 445 | self._abort(request_id) |
| 446 | raise e |
| 447 | |
| 448 | async def abort(self, request_id: str) -> None: |
| 449 | """Abort a request. |
nothing calls this directly
no test coverage detected