11import time
22import types
33from abc import ABC
4+ from pathlib import Path
45from typing import Annotated , Optional , Type , overload
56
67from dotenv import load_dotenv
1617from askui .tools .agent_os import AgentOs
1718from askui .tools .android .agent_os import AndroidAgentOs
1819from askui .utils .image_utils import ImageSource , Img
20+ from askui .utils .pdf_utils import Pdf
21+ from askui .utils .source_utils import load_image_source , load_source
1922
2023from .logger import configure_logging , logger
2124from .models import ModelComposition
@@ -190,46 +193,53 @@ def get(
190193 query : Annotated [str , Field (min_length = 1 )],
191194 response_schema : None = None ,
192195 model : str | None = None ,
193- image : Optional [Img ] = None ,
196+ source : Optional [Img | Pdf ] = None ,
194197 ) -> str : ...
195198 @overload
196199 def get (
197200 self ,
198201 query : Annotated [str , Field (min_length = 1 )],
199202 response_schema : Type [ResponseSchema ],
200203 model : str | None = None ,
201- image : Optional [Img ] = None ,
204+ source : Optional [Img | Pdf ] = None ,
202205 ) -> ResponseSchema : ...
203206
204- @telemetry .record_call (exclude = {"query" , "image " , "response_schema" })
207+ @telemetry .record_call (exclude = {"query" , "source " , "response_schema" })
205208 @validate_call (config = ConfigDict (arbitrary_types_allowed = True ))
206209 def get (
207210 self ,
208211 query : Annotated [str , Field (min_length = 1 )],
209212 response_schema : Type [ResponseSchema ] | None = None ,
210213 model : str | None = None ,
211- image : Optional [Img ] = None ,
214+ source : Optional [Img | Pdf ] = None ,
212215 ) -> ResponseSchema | str :
213216 """
214- Retrieves information from an image (defaults to a screenshot of the current
215- screen) based on the provided `query`.
217+ Retrieves information from an image or PDF based on the provided `query`.
218+
219+ If no `source` is provided, a screenshot of the current screen is taken.
216220
217221 Args:
218222 query (str): The query describing what information to retrieve.
219- image (Img | None, optional): The image to extract information from.
220- Defaults to a screenshot of the current screen. Can be a path to
221- an image file, a PIL Image object or a data URL .
223+ source (Img | Pdf | None, optional): The source to extract information from.
224+ Can be a path to a PDF file, a path to an image file, a PIL Image
225+ object or a data URL. Defaults to a screenshot of the current screen .
222226 response_schema (Type[ResponseSchema] | None, optional): A Pydantic model
223227 class that defines the response schema. If not provided, returns a
224228 string.
225229 model (str | None, optional): The composition or name of the model(s) to
226230 be used for retrieving information from the screen or image using the
227231 `query`. Note: `response_schema` is not supported by all models.
232+ PDF processing is only supported for Gemini models hosted on AskUI.
228233
229234 Returns:
230235 ResponseSchema | str: The extracted information, `str` if no
231236 `response_schema` is provided.
232237
238+ Raises:
239+ NotImplementedError: If PDF processing is not supported for the selected
240+ model.
241+ ValueError: If the `source` is not a valid PDF or image.
242+
233243 Example:
234244 ```python
235245 from askui import ResponseSchemaBase, VisionAgent
@@ -254,7 +264,7 @@ class LinkedListNode(ResponseSchemaBase):
254264 response = agent.get(
255265 "What is the current url shown in the url bar?",
256266 response_schema=UrlResponse,
257- image ="screenshot.png",
267+ source ="screenshot.png",
258268 )
259269 # Dump whole model
260270 print(response.model_dump_json(indent=2))
@@ -269,7 +279,7 @@ class LinkedListNode(ResponseSchemaBase):
269279 is_login_page = agent.get(
270280 "Is this a login page?",
271281 response_schema=bool,
272- image =Image.open("screenshot.png"),
282+ source =Image.open("screenshot.png"),
273283 )
274284 print(is_login_page)
275285
@@ -303,13 +313,34 @@ class LinkedListNode(ResponseSchemaBase):
303313 while current:
304314 print(current.value)
305315 current = current.next
316+
317+ # Get text from PDF
318+ text = agent.get(
319+ "Extract all text from the PDF",
320+ source="document.pdf",
321+ )
322+ print(text)
306323 ```
307324 """
308325 logger .debug ("VisionAgent received instruction to get '%s'" , query )
309- _image = ImageSource (self ._agent_os .screenshot () if image is None else image )
310- self ._reporter .add_message ("User" , f'get: "{ query } "' , image = _image .root )
326+ _source = (
327+ ImageSource (self ._agent_os .screenshot ())
328+ if source is None
329+ else load_source (source )
330+ )
331+
332+ # Prepare message content with file path if available
333+ user_message_content = f'get: "{ query } "' + (
334+ f" from '{ source } '" if isinstance (source , (str , Path )) else ""
335+ )
336+
337+ self ._reporter .add_message (
338+ "User" ,
339+ user_message_content ,
340+ image = _source .root if isinstance (_source , ImageSource ) else None ,
341+ )
311342 response = self ._model_router .get (
312- image = _image ,
343+ source = _source ,
313344 query = query ,
314345 response_schema = response_schema ,
315346 model_choice = model or self ._model_choice ["get" ],
@@ -329,8 +360,8 @@ def _locate(
329360 screenshot : Optional [Img ] = None ,
330361 model : ModelComposition | str | None = None ,
331362 ) -> PointList :
332- def locate_with_screenshot () -> PointList :
333- _screenshot = ImageSource (
363+ def locate_with_screenshot () -> Point :
364+ _screenshot = load_image_source (
334365 self ._agent_os .screenshot () if screenshot is None else screenshot
335366 )
336367 return self ._model_router .locate (
0 commit comments