Shared CLI arguments for vLLM engine.
(
parser: argparse.ArgumentParser)
| 41 | |
| 42 | @staticmethod |
| 43 | def add_cli_args( |
| 44 | parser: argparse.ArgumentParser) -> argparse.ArgumentParser: |
| 45 | """Shared CLI arguments for vLLM engine.""" |
| 46 | |
| 47 | # NOTE: If you update any of the arguments below, please also |
| 48 | # make sure to update docs/source/models/engine_args.rst |
| 49 | |
| 50 | # Model arguments |
| 51 | parser.add_argument( |
| 52 | '--model', |
| 53 | type=str, |
| 54 | default='facebook/opt-125m', |
| 55 | help='name or path of the huggingface model to use') |
| 56 | parser.add_argument( |
| 57 | '--tokenizer', |
| 58 | type=str, |
| 59 | default=EngineArgs.tokenizer, |
| 60 | help='name or path of the huggingface tokenizer to use') |
| 61 | parser.add_argument( |
| 62 | '--revision', |
| 63 | type=str, |
| 64 | default=None, |
| 65 | help='the specific model version to use. It can be a branch ' |
| 66 | 'name, a tag name, or a commit id. If unspecified, will use ' |
| 67 | 'the default version.') |
| 68 | parser.add_argument( |
| 69 | '--tokenizer-revision', |
| 70 | type=str, |
| 71 | default=None, |
| 72 | help='the specific tokenizer version to use. It can be a branch ' |
| 73 | 'name, a tag name, or a commit id. If unspecified, will use ' |
| 74 | 'the default version.') |
| 75 | parser.add_argument('--tokenizer-mode', |
| 76 | type=str, |
| 77 | default=EngineArgs.tokenizer_mode, |
| 78 | choices=['auto', 'slow'], |
| 79 | help='tokenizer mode. "auto" will use the fast ' |
| 80 | 'tokenizer if available, and "slow" will ' |
| 81 | 'always use the slow tokenizer.') |
| 82 | parser.add_argument('--trust-remote-code', |
| 83 | action='store_true', |
| 84 | help='trust remote code from huggingface') |
| 85 | parser.add_argument('--download-dir', |
| 86 | type=str, |
| 87 | default=EngineArgs.download_dir, |
| 88 | help='directory to download and load the weights, ' |
| 89 | 'default to the default cache dir of ' |
| 90 | 'huggingface') |
| 91 | parser.add_argument( |
| 92 | '--load-format', |
| 93 | type=str, |
| 94 | default=EngineArgs.load_format, |
| 95 | choices=['auto', 'pt', 'safetensors', 'npcache', 'dummy'], |
| 96 | help='The format of the model weights to load. ' |
| 97 | '"auto" will try to load the weights in the safetensors format ' |
| 98 | 'and fall back to the pytorch bin format if safetensors format ' |
| 99 | 'is not available. ' |
| 100 | '"pt" will load the weights in the pytorch bin format. ' |
no outgoing calls
no test coverage detected