INNER CODE UNIT · Python

find_element

instavm/clickclickclick · clickclickclick/finder/__init__.py:52

    def find_element(self, prompt, observation: str) -> str:
        new_size = self.IMAGE_WIDTH  # assuming square image size
        logger.info(prompt)
        screenshot = self.executor.screenshot(observation, False, True)
        image = Image.open(screenshot)

        segments, total_width, total_height = self.resize(image, new_size=new_size)

        results = [self.process_segment(segments[0], self.model_name, prompt)]
        i = 0
        ans = "0,0,0,0"
        for response, coordinates in results:
            i += 1
            print(coordinates, i)

            try:
                # Try parsing as JSON first (for models that return JSON)
                response_dict = json.loads(response)

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…