(
self,
images: ImageInput = None,
text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]] = None,
unnorm_key: Optional[str] = None,
suffix_actions: Optional[np.array] = None, # (t e)
**kwargs: Unpack[PaliGemmaProcessorKwargs],
)
| 101 | ) |
| 102 | |
| 103 | def __call__( |
| 104 | self, |
| 105 | images: ImageInput = None, |
| 106 | text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]] = None, |
| 107 | unnorm_key: Optional[str] = None, |
| 108 | suffix_actions: Optional[np.array] = None, # (t e) |
| 109 | **kwargs: Unpack[PaliGemmaProcessorKwargs], |
| 110 | ) -> BatchFeature: |
| 111 | images, text = _validate_images_text_input_order(images, text) |
| 112 | |
| 113 | output_kwargs = self._merge_kwargs( |
| 114 | PaliGemmaProcessorKwargs, |
| 115 | tokenizer_init_kwargs=self.tokenizer.init_kwargs, |
| 116 | **kwargs, |
| 117 | ) |
| 118 | if suffix_actions is not None: |
| 119 | action_tokens = self.action_tokenizer(suffix_actions) # (n,3) |
| 120 | suffix="".join(action_tokens.flatten()) |
| 121 | else: |
| 122 | suffix = output_kwargs["text_kwargs"].pop("suffix", None) |
| 123 | |
| 124 | return_token_type_ids = True if suffix is not None else False |
| 125 | |
| 126 | if images is None: |
| 127 | raise ValueError("`images` are expected as arguments to a `PaliGemmaProcessor` instance.") |
| 128 | if text is None: |
| 129 | logger.warning_once( "You are using PaliGemma without a text prefix. It will perform as a picture-captioning model.") |
| 130 | text = "" |
| 131 | |
| 132 | if _is_str_or_image(text): |
| 133 | text = [text] |
| 134 | elif isinstance(text, list) and _is_str_or_image(text[0]): |
| 135 | pass |
| 136 | |
| 137 | if text is not None and images is not None: |
| 138 | if not any(IMAGE_TOKEN in sample for sample in text): |
| 139 | if isinstance(text, List) and isinstance(images, List): |
| 140 | if len(images) != len(text): |
| 141 | raise ValueError( |
| 142 | f"Received {len(images)} images for {len(text)} prompts. Each prompt should be associated with an image or list of images." |
| 143 | ) |
| 144 | if is_valid_image(images): |
| 145 | images = [[images]] |
| 146 | elif isinstance(images, list) and is_valid_image(images[0]): |
| 147 | images = [[image] for image in images] |
| 148 | elif not (isinstance(images, list) and isinstance(images[0], list) and is_valid_image(images[0][0])): |
| 149 | raise ValueError("images must be an image, list of images or list of list of images") |
| 150 | if suffix is not None and _is_str_or_image(suffix): suffix = [suffix] |
| 151 | if suffix is not None: suffix = [sfx + self.tokenizer.eos_token for sfx in suffix] |
| 152 | input_strings = [ |
| 153 | build_string_from_input( |
| 154 | prompt=prompt, |
| 155 | bos_token=self.tokenizer.bos_token, |
| 156 | image_seq_len=self.image_seq_length, |
| 157 | image_token=IMAGE_TOKEN, |
| 158 | num_images=len(image_list) if isinstance(image_list, list) else 1, |
| 159 | ) |
| 160 | for prompt, image_list in zip(text, images) |
nothing calls this directly
no outgoing calls
no test coverage detected