Skip to content

Commit 0aff01f

Browse files
Merge branch 'main' into feat-add-locate-all
2 parents 748b987 + cf0a4f5 commit 0aff01f

32 files changed

Lines changed: 612 additions & 250 deletions

‎README.md‎

Lines changed: 11 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -350,7 +350,7 @@ class MyGetAndLocateModel(GetModel, LocateModel):
350350
def get(
351351
self,
352352
query: str,
353-
image: ImageSource,
353+
source: Source,
354354
response_schema: Type[ResponseSchema] | None,
355355
model_choice: str,
356356
) -> ResponseSchema | str:
@@ -640,9 +640,9 @@ else:
640640
agent.click("Login")
641641
```
642642

643-
#### Using custom images
643+
#### Using custom images and PDFs
644644

645-
Instead of taking a screenshot, you can analyze specific images:
645+
Instead of taking a screenshot, you can analyze specific images or PDFs:
646646

647647
```python
648648
from PIL import Image
@@ -651,10 +651,13 @@ from askui import VisionAgent
651651
# From PIL Image
652652
with VisionAgent() as agent:
653653
image = Image.open("screenshot.png")
654-
result = agent.get("What's in this image?", image)
654+
result = agent.get("What's in this image?", source=image)
655655

656656
# From file path
657-
result = agent.get("What's in this image?", "screenshot.png")
657+
result = agent.get("What's in this image?", source="screenshot.png")
658+
659+
# From PDF
660+
result = agent.get("What is this PDF about?", source="document.pdf")
658661
```
659662

660663
#### Using response schemas
@@ -696,7 +699,7 @@ with VisionAgent() as agent:
696699
response = agent.get(
697700
"What is the current url shown in the url bar?",
698701
response_schema=UrlResponse,
699-
image="screenshot.png",
702+
source="screenshot.png",
700703
)
701704

702705
# Dump whole model
@@ -712,7 +715,7 @@ with VisionAgent() as agent:
712715
is_login_page = agent.get(
713716
"Is this a login page?",
714717
response_schema=bool,
715-
image=Image.open("screenshot.png"),
718+
source=Image.open("screenshot.png"),
716719
)
717720
print(is_login_page)
718721

@@ -751,6 +754,7 @@ with VisionAgent() as agent:
751754
**⚠️ Limitations:**
752755
- The support for response schemas varies among models. Currently, the `askui` model provides best support for response schemas
753756
as we try different models under the hood with your schema to see which one works best.
757+
- PDF processing is only supported for Gemini models hosted on AskUI and for PDFs up to 20MB.
754758

755759
## What is AskUI Vision Agent?
756760

‎pdm.lock‎

Lines changed: 11 additions & 1 deletion
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

‎pyproject.toml‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -25,6 +25,7 @@ dependencies = [
2525
"jsonref>=1.1.0",
2626
"protobuf>=6.31.1",
2727
"google-genai>=1.20.0",
28+
"filetype>=1.2.0",
2829
]
2930
requires-python = ">=3.10"
3031
readme = "README.md"

‎src/askui/__init__.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
"""AskUI Vision Agent"""
22

3-
__version__ = "0.11.0"
3+
__version__ = "0.12.0"
44

55
from .agent import VisionAgent
66
from .locators import Locator

‎src/askui/agent.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -48,15 +48,15 @@
4848

4949
_ANTHROPIC__CLAUDE__3_5__SONNET__20241022__ACT_SETTINGS = ActSettings(
5050
messages=MessageSettings(
51-
model=ModelName.ANTHROPIC__CLAUDE__3_5__SONNET__20241022.value,
51+
model=ModelName.ANTHROPIC__CLAUDE__3_5__SONNET__20241022,
5252
system=_SYSTEM_PROMPT,
5353
betas=[COMPUTER_USE_20241022_BETA_FLAG],
5454
),
5555
)
5656

5757
_CLAUDE__SONNET__4__20250514__ACT_SETTINGS = ActSettings(
5858
messages=MessageSettings(
59-
model=ModelName.CLAUDE__SONNET__4__20250514.value,
59+
model=ModelName.CLAUDE__SONNET__4__20250514,
6060
system=_SYSTEM_PROMPT,
6161
betas=[COMPUTER_USE_20250124_BETA_FLAG],
6262
thinking={"type": "enabled", "budget_tokens": 2048},

‎src/askui/agent_base.py‎

Lines changed: 47 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,7 @@
11
import time
22
import types
33
from abc import ABC
4+
from pathlib import Path
45
from typing import Annotated, Optional, Type, overload
56

67
from dotenv import load_dotenv
@@ -16,6 +17,8 @@
1617
from askui.tools.agent_os import AgentOs
1718
from askui.tools.android.agent_os import AndroidAgentOs
1819
from askui.utils.image_utils import ImageSource, Img
20+
from askui.utils.pdf_utils import Pdf
21+
from askui.utils.source_utils import load_image_source, load_source
1922

2023
from .logger import configure_logging, logger
2124
from .models import ModelComposition
@@ -190,46 +193,53 @@ def get(
190193
query: Annotated[str, Field(min_length=1)],
191194
response_schema: None = None,
192195
model: str | None = None,
193-
image: Optional[Img] = None,
196+
source: Optional[Img | Pdf] = None,
194197
) -> str: ...
195198
@overload
196199
def get(
197200
self,
198201
query: Annotated[str, Field(min_length=1)],
199202
response_schema: Type[ResponseSchema],
200203
model: str | None = None,
201-
image: Optional[Img] = None,
204+
source: Optional[Img | Pdf] = None,
202205
) -> ResponseSchema: ...
203206

204-
@telemetry.record_call(exclude={"query", "image", "response_schema"})
207+
@telemetry.record_call(exclude={"query", "source", "response_schema"})
205208
@validate_call(config=ConfigDict(arbitrary_types_allowed=True))
206209
def get(
207210
self,
208211
query: Annotated[str, Field(min_length=1)],
209212
response_schema: Type[ResponseSchema] | None = None,
210213
model: str | None = None,
211-
image: Optional[Img] = None,
214+
source: Optional[Img | Pdf] = None,
212215
) -> ResponseSchema | str:
213216
"""
214-
Retrieves information from an image (defaults to a screenshot of the current
215-
screen) based on the provided `query`.
217+
Retrieves information from an image or PDF based on the provided `query`.
218+
219+
If no `source` is provided, a screenshot of the current screen is taken.
216220
217221
Args:
218222
query (str): The query describing what information to retrieve.
219-
image (Img | None, optional): The image to extract information from.
220-
Defaults to a screenshot of the current screen. Can be a path to
221-
an image file, a PIL Image object or a data URL.
223+
source (Img | Pdf | None, optional): The source to extract information from.
224+
Can be a path to a PDF file, a path to an image file, a PIL Image
225+
object or a data URL. Defaults to a screenshot of the current screen.
222226
response_schema (Type[ResponseSchema] | None, optional): A Pydantic model
223227
class that defines the response schema. If not provided, returns a
224228
string.
225229
model (str | None, optional): The composition or name of the model(s) to
226230
be used for retrieving information from the screen or image using the
227231
`query`. Note: `response_schema` is not supported by all models.
232+
PDF processing is only supported for Gemini models hosted on AskUI.
228233
229234
Returns:
230235
ResponseSchema | str: The extracted information, `str` if no
231236
`response_schema` is provided.
232237
238+
Raises:
239+
NotImplementedError: If PDF processing is not supported for the selected
240+
model.
241+
ValueError: If the `source` is not a valid PDF or image.
242+
233243
Example:
234244
```python
235245
from askui import ResponseSchemaBase, VisionAgent
@@ -254,7 +264,7 @@ class LinkedListNode(ResponseSchemaBase):
254264
response = agent.get(
255265
"What is the current url shown in the url bar?",
256266
response_schema=UrlResponse,
257-
image="screenshot.png",
267+
source="screenshot.png",
258268
)
259269
# Dump whole model
260270
print(response.model_dump_json(indent=2))
@@ -269,7 +279,7 @@ class LinkedListNode(ResponseSchemaBase):
269279
is_login_page = agent.get(
270280
"Is this a login page?",
271281
response_schema=bool,
272-
image=Image.open("screenshot.png"),
282+
source=Image.open("screenshot.png"),
273283
)
274284
print(is_login_page)
275285
@@ -303,13 +313,34 @@ class LinkedListNode(ResponseSchemaBase):
303313
while current:
304314
print(current.value)
305315
current = current.next
316+
317+
# Get text from PDF
318+
text = agent.get(
319+
"Extract all text from the PDF",
320+
source="document.pdf",
321+
)
322+
print(text)
306323
```
307324
"""
308325
logger.debug("VisionAgent received instruction to get '%s'", query)
309-
_image = ImageSource(self._agent_os.screenshot() if image is None else image)
310-
self._reporter.add_message("User", f'get: "{query}"', image=_image.root)
326+
_source = (
327+
ImageSource(self._agent_os.screenshot())
328+
if source is None
329+
else load_source(source)
330+
)
331+
332+
# Prepare message content with file path if available
333+
user_message_content = f'get: "{query}"' + (
334+
f" from '{source}'" if isinstance(source, (str, Path)) else ""
335+
)
336+
337+
self._reporter.add_message(
338+
"User",
339+
user_message_content,
340+
image=_source.root if isinstance(_source, ImageSource) else None,
341+
)
311342
response = self._model_router.get(
312-
image=_image,
343+
source=_source,
313344
query=query,
314345
response_schema=response_schema,
315346
model_choice=model or self._model_choice["get"],
@@ -329,8 +360,8 @@ def _locate(
329360
screenshot: Optional[Img] = None,
330361
model: ModelComposition | str | None = None,
331362
) -> PointList:
332-
def locate_with_screenshot() -> PointList:
333-
_screenshot = ImageSource(
363+
def locate_with_screenshot() -> Point:
364+
_screenshot = load_image_source(
334365
self._agent_os.screenshot() if screenshot is None else screenshot
335366
)
336367
return self._model_router.locate(

‎src/askui/android_agent.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -103,15 +103,15 @@
103103

104104
_ANTHROPIC__CLAUDE__3_5__SONNET__20241022__ACT_SETTINGS = ActSettings(
105105
messages=MessageSettings(
106-
model=ModelName.ANTHROPIC__CLAUDE__3_5__SONNET__20241022.value,
106+
model=ModelName.ANTHROPIC__CLAUDE__3_5__SONNET__20241022,
107107
system=_SYSTEM_PROMPT,
108108
betas=[],
109109
),
110110
)
111111

112112
_CLAUDE__SONNET__4__20250514__ACT_SETTINGS = ActSettings(
113113
messages=MessageSettings(
114-
model=ModelName.CLAUDE__SONNET__4__20250514.value,
114+
model=ModelName.CLAUDE__SONNET__4__20250514,
115115
system=_SYSTEM_PROMPT,
116116
thinking={"type": "enabled", "budget_tokens": 2048},
117117
betas=[],

‎src/askui/chat/api/messages/service.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -64,7 +64,7 @@ def list_(self, thread_id: ThreadId, query: ListQuery) -> list[Message]:
6464
raise FileNotFoundError(error_msg)
6565

6666
messages: list[Message] = []
67-
with thread_file.open("r") as f:
67+
with thread_file.open("r", encoding="utf-8") as f:
6868
for line in f:
6969
msg = Message.model_validate_json(line)
7070
messages.append(msg)

‎src/askui/locators/locators.py‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -7,7 +7,7 @@
77
from pydantic import ConfigDict, Field, validate_call
88

99
from askui.locators.relatable import Relatable
10-
from askui.utils.image_utils import ImageSource
10+
from askui.utils.source_utils import load_image_source
1111

1212
TextMatchType = Literal["similar", "exact", "contains", "regex"]
1313
"""The type of match to use.
@@ -303,7 +303,7 @@ def __init__(
303303
image_compare_format=image_compare_format,
304304
name=_generate_name() if name is None else name,
305305
)
306-
self._image = ImageSource(image)
306+
self._image = load_image_source(image)
307307

308308

309309
class AiElement(ImageBase):

‎src/askui/models/anthropic/messages_api.py‎

Lines changed: 7 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -48,6 +48,8 @@
4848
scale_coordinates,
4949
scale_image_to_fit,
5050
)
51+
from askui.utils.pdf_utils import PdfSource
52+
from askui.utils.source_utils import Source
5153

5254
from .utils import extract_click_coordinates
5355

@@ -63,10 +65,6 @@ def build_system_prompt_locate(screen_width: str, screen_height: str) -> str:
6365
"claude-3-5-sonnet-20241022"
6466
),
6567
ModelName.CLAUDE__SONNET__4__20250514: "claude-sonnet-4-20250514",
66-
ModelName.ANTHROPIC__CLAUDE__3_5__SONNET__20241022.value: (
67-
"claude-3-5-sonnet-20241022"
68-
),
69-
ModelName.CLAUDE__SONNET__4__20250514.value: "claude-sonnet-4-20250514",
7068
}
7169
)
7270
)
@@ -240,16 +238,19 @@ def locate(
240238
def get(
241239
self,
242240
query: str,
243-
image: ImageSource,
241+
source: Source,
244242
response_schema: Type[ResponseSchema] | None,
245243
model_choice: str,
246244
) -> ResponseSchema | str:
245+
if isinstance(source, PdfSource):
246+
err_msg = f"PDF processing is not supported for the model {model_choice}"
247+
raise NotImplementedError(err_msg)
247248
try:
248249
if response_schema is not None:
249250
error_msg = "Response schema is not yet supported for Anthropic"
250251
raise NotImplementedError(error_msg)
251252
return self._inference(
252-
image=image,
253+
image=source,
253254
prompt=query,
254255
system=SYSTEM_PROMPT_GET,
255256
model_choice=model_choice,

0 commit comments

Comments
 (0)