INNER CODE UNIT · Python
find_element
instavm/clickclickclick · clickclickclick/finder/__init__.py:52
def find_element(self, prompt, observation: str) -> str:
new_size = self.IMAGE_WIDTH # assuming square image size
logger.info(prompt)
screenshot = self.executor.screenshot(observation, False, True)
image = Image.open(screenshot)
segments, total_width, total_height = self.resize(image, new_size=new_size)
results = [self.process_segment(segments[0], self.model_name, prompt)]
i = 0
ans = "0,0,0,0"
for response, coordinates in results:
i += 1
print(coordinates, i)
try:
# Try parsing as JSON first (for models that return JSON)
response_dict = json.loads(response)