MCPcopy Create free account
hub / github.com/SpatialVLA/SpatialVLA / __call__

Method __call__

model/processing_spatialvla.py:103–192  ·  view source on GitHub ↗
(
        self,
        images: ImageInput = None,
        text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]] = None,
        unnorm_key: Optional[str] = None,
        suffix_actions: Optional[np.array] = None, # (t e)
        **kwargs: Unpack[PaliGemmaProcessorKwargs],
    )

Source from the content-addressed store, hash-verified

101 )
102
103 def __call__(
104 self,
105 images: ImageInput = None,
106 text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]] = None,
107 unnorm_key: Optional[str] = None,
108 suffix_actions: Optional[np.array] = None, # (t e)
109 **kwargs: Unpack[PaliGemmaProcessorKwargs],
110 ) -> BatchFeature:
111 images, text = _validate_images_text_input_order(images, text)
112
113 output_kwargs = self._merge_kwargs(
114 PaliGemmaProcessorKwargs,
115 tokenizer_init_kwargs=self.tokenizer.init_kwargs,
116 **kwargs,
117 )
118 if suffix_actions is not None:
119 action_tokens = self.action_tokenizer(suffix_actions) # (n,3)
120 suffix="".join(action_tokens.flatten())
121 else:
122 suffix = output_kwargs["text_kwargs"].pop("suffix", None)
123
124 return_token_type_ids = True if suffix is not None else False
125
126 if images is None:
127 raise ValueError("`images` are expected as arguments to a `PaliGemmaProcessor` instance.")
128 if text is None:
129 logger.warning_once( "You are using PaliGemma without a text prefix. It will perform as a picture-captioning model.")
130 text = ""
131
132 if _is_str_or_image(text):
133 text = [text]
134 elif isinstance(text, list) and _is_str_or_image(text[0]):
135 pass
136
137 if text is not None and images is not None:
138 if not any(IMAGE_TOKEN in sample for sample in text):
139 if isinstance(text, List) and isinstance(images, List):
140 if len(images) != len(text):
141 raise ValueError(
142 f"Received {len(images)} images for {len(text)} prompts. Each prompt should be associated with an image or list of images."
143 )
144 if is_valid_image(images):
145 images = [[images]]
146 elif isinstance(images, list) and is_valid_image(images[0]):
147 images = [[image] for image in images]
148 elif not (isinstance(images, list) and isinstance(images[0], list) and is_valid_image(images[0][0])):
149 raise ValueError("images must be an image, list of images or list of list of images")
150 if suffix is not None and _is_str_or_image(suffix): suffix = [suffix]
151 if suffix is not None: suffix = [sfx + self.tokenizer.eos_token for sfx in suffix]
152 input_strings = [
153 build_string_from_input(
154 prompt=prompt,
155 bos_token=self.tokenizer.bos_token,
156 image_seq_len=self.image_seq_length,
157 image_token=IMAGE_TOKEN,
158 num_images=len(image_list) if isinstance(image_list, list) else 1,
159 )
160 for prompt, image_list in zip(text, images)

Callers

nothing calls this directly

Calls

no outgoing calls

Tested by

no test coverage detected