Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 16 additions & 9 deletions libs/partners/decodo/Makefile
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@

.PHONY: all lint format test tests integration_tests help
.PHONY: all lint lint_package lint_tests format test tests integration_tests help

PYTEST_EXTRA ?=

all: help

Expand All @@ -8,24 +10,29 @@ all: help
######################

lint format: PYTHON_FILES=.
lint:
ruff check $(PYTHON_FILES)
ruff format $(PYTHON_FILES) --diff
mypy $(PYTHON_FILES)
lint_package: PYTHON_FILES=langchain_decodo
lint_tests: PYTHON_FILES=tests
lint_tests: MYPY_CACHE=.mypy_cache_test
MYPY_CACHE=.mypy_cache

lint lint_package lint_tests:
uv run --group lint --group typing ruff check $(PYTHON_FILES)
uv run --group lint --group typing ruff format $(PYTHON_FILES) --diff
mkdir -p $(MYPY_CACHE) && uv run --group lint --group typing mypy $(PYTHON_FILES) --cache-dir $(MYPY_CACHE)

format:
ruff format $(PYTHON_FILES)
ruff check --fix $(PYTHON_FILES)
uv run --group lint ruff format $(PYTHON_FILES)
uv run --group lint ruff check --fix $(PYTHON_FILES)

######################
# TESTING
######################

tests test:
pytest tests/unit_tests
uv run --group test pytest $(PYTEST_EXTRA) tests/unit_tests

integration_tests:
pytest tests/integration_tests
uv run --group test --group test_integration pytest tests/integration_tests

######################
# HELP
Expand Down
85 changes: 52 additions & 33 deletions libs/partners/decodo/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,14 @@ export DECODO_API_TOKEN="your-decodo-api-token"

Get a token from the [Decodo Dashboard](https://app.decodo.com).

By default every class expects basic credentials (base64-encoded
`username:password`). If your token is a plain API token, pass
`auth_mode="token"`.:

```python
tool = DecodoWebScrapeTool(auth_mode="token")
```

## Components

### `DecodoWebScrapeTool`
Expand Down Expand Up @@ -71,11 +79,11 @@ Returns a JSON string — a list of objects with `content`, `url`, and

Supported engines:

| `engine` | Decodo target | Description |
|---|---|---|
| `google` | `google_search` | Google SERP |
| `amazon` | `amazon_search` | Amazon product search |
| `reddit` | `google_search` + `site:reddit.com` | Reddit via Google |
| `engine` | Decodo target | Description |
| -------- | ----------------------------------- | --------------------- |
| `google` | `google_search` | Google SERP |
| `amazon` | `amazon_search` | Amazon product search |
| `reddit` | `google_search` + `site:reddit.com` | Reddit via Google |

### `DecodoLoader`

Expand Down Expand Up @@ -105,53 +113,64 @@ Each `Document` has:

## LangChain agent example

```bash
pip install langchain langchain-openai langchain-decodo
```

```python
from langchain import hub
from langchain.agents import AgentExecutor, create_react_agent
from langchain_openai import ChatOpenAI
from langchain_decodo import DecodoWebScrapeTool, DecodoSearchTool

tools = [DecodoWebScrapeTool(), DecodoSearchTool()]
llm = ChatOpenAI(model="gpt-4o-mini", temperature=0)
prompt = hub.pull("hwchase17/react")

agent = create_react_agent(llm=llm, tools=tools, prompt=prompt)
executor = AgentExecutor(agent=agent, tools=tools, verbose=True)

result = executor.invoke({
"input": "What is the latest stable version of Python? Check python.org."
})
print(result["output"])
from langchain.agents import create_agent
from langchain_decodo import DecodoSearchTool, DecodoWebScrapeTool

agent = create_agent(
model="openai:gpt-4o-mini",
tools=[DecodoWebScrapeTool(), DecodoSearchTool()],
)

result = agent.invoke(
{
"messages": [
{
"role": "user",
"content": "What is the latest stable version of Python? Check python.org.",
}
]
}
)
print(result["messages"][-1].content)
```

## RAG pipeline example

```bash
pip install langchain-openai langchain-text-splitters langchain-decodo numpy
```

```python
from langchain.text_splitter import RecursiveCharacterTextSplitter
from langchain_community.vectorstores import FAISS
from langchain_core.vectorstores import InMemoryVectorStore
from langchain_openai import ChatOpenAI, OpenAIEmbeddings
from langchain.chains import RetrievalQA
from langchain_text_splitters import RecursiveCharacterTextSplitter
from langchain_decodo import DecodoLoader

loader = DecodoLoader(urls=["https://python.org/about/"])
docs = loader.load()
docs = DecodoLoader(urls=["https://python.org/about/"]).load()

splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
chunks = splitter.split_documents(docs)

store = FAISS.from_documents(chunks, OpenAIEmbeddings())
chain = RetrievalQA.from_chain_type(
llm=ChatOpenAI(model="gpt-4o-mini"),
retriever=store.as_retriever(search_kwargs={"k": 4}),
store = InMemoryVectorStore.from_documents(chunks, OpenAIEmbeddings())
context = "\n\n".join(
doc.page_content for doc in store.similarity_search("What is Python used for?", k=4)
)

result = chain.invoke({"query": "What is Python used for?"})
print(result["result"])
llm = ChatOpenAI(model="gpt-4o-mini")
answer = llm.invoke(
f"Answer using only this context:\n\n{context}\n\nQuestion: What is Python used for?"
)
print(answer.content)
```

## Links

- [Decodo website](https://decodo.com)
- [Decodo API documentation](https://developers.decodo.com)
- [Decodo API documentation](https://help.decodo.com)
- [Decodo Dashboard](https://app.decodo.com)
- [LangChain documentation](https://python.langchain.com)
2 changes: 1 addition & 1 deletion libs/partners/decodo/langchain_decodo/_version.py
Original file line number Diff line number Diff line change
@@ -1 +1 @@
__version__ = "0.1.0"
__version__ = "0.1.1"
17 changes: 9 additions & 8 deletions libs/partners/decodo/langchain_decodo/document_loaders.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,8 @@

import json
import os
from typing import Any, Iterator, Literal
from collections.abc import Iterator
from typing import Any, Literal

import httpx
from langchain_core.document_loaders import BaseLoader
Expand Down Expand Up @@ -103,7 +104,11 @@ def _scrape_url(
RuntimeError: On timeout, network error, or non-2xx HTTP response.
"""
endpoint = f"{_API_BASE}{_scrape_path(auth_mode)}"
payload: dict[str, Any] = {"target": "universal", "url": url}
payload: dict[str, Any] = {
"target": "universal",
"url": url,
"markdown": True,
}

try:
response = httpx.post(
Expand All @@ -113,13 +118,9 @@ def _scrape_url(
timeout=timeout,
)
except httpx.TimeoutException as exc:
raise RuntimeError(
f"Decodo API request for '{url}' timed out after {timeout}s"
) from exc
raise RuntimeError(f"Decodo API request for '{url}' timed out after {timeout}s") from exc
except httpx.RequestError as exc:
raise RuntimeError(
f"Decodo API network error while fetching '{url}': {exc}"
) from exc
raise RuntimeError(f"Decodo API network error while fetching '{url}': {exc}") from exc

if not response.is_success:
try:
Expand Down
43 changes: 28 additions & 15 deletions libs/partners/decodo/langchain_decodo/tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@

import json
import os
from typing import Any, Literal, Optional, Type
from typing import Any, Literal

import httpx
from langchain_core.callbacks import CallbackManagerForToolRun
Expand Down Expand Up @@ -105,9 +105,7 @@ def _do_scrape(
try:
response = httpx.post(url, headers=headers, json=payload, timeout=timeout)
except httpx.TimeoutException as exc:
raise RuntimeError(
f"Decodo API request timed out after {timeout}s: {exc}"
) from exc
raise RuntimeError(f"Decodo API request timed out after {timeout}s: {exc}") from exc
except httpx.RequestError as exc:
raise RuntimeError(f"Decodo API network error: {exc}") from exc

Expand All @@ -123,6 +121,12 @@ def _do_scrape(
return response.json() # type: ignore[no-any-return]


def _is_failed(entry: dict[str, Any]) -> bool:
"""Return whether a result entry carries a failed (4xx/5xx or 6xx) status."""
status = entry.get("status_code")
return isinstance(status, int) and status >= 400


def _extract_content(response: dict[str, Any]) -> str:
"""Pull the first result's content string from a Decodo API response.

Expand Down Expand Up @@ -225,7 +229,7 @@ class DecodoWebScrapeTool(BaseTool):
"automatically. Use this when you need the complete text of a specific URL. "
"Input: a valid URL string (must include http:// or https://)."
)
args_schema: Type[BaseModel] = _WebScrapeInput
args_schema: type[BaseModel] = _WebScrapeInput

decodo_api_token: SecretStr = Field(
default=SecretStr(""),
Expand Down Expand Up @@ -267,7 +271,7 @@ def validate_api_token(cls, values: dict[str, Any]) -> dict[str, Any]:
def _run(
self,
url: str,
run_manager: Optional[CallbackManagerForToolRun] = None,
run_manager: CallbackManagerForToolRun | None = None,
) -> str:
"""Scrape the given URL and return its content.

Expand All @@ -289,7 +293,11 @@ def _run(
"field or the ``DECODO_API_TOKEN`` environment variable."
)

payload: dict[str, Any] = {"target": "universal", "url": url}
payload: dict[str, Any] = {
"target": "universal",
"url": url,
"markdown": True,
}
response = _do_scrape(token, self.base_url, payload, auth_mode=self.auth_mode)
content = _extract_content(response)
return content if content else "(No content returned by Decodo API)"
Expand Down Expand Up @@ -344,7 +352,7 @@ class DecodoSearchTool(BaseTool):
"'num_results' (optional: integer 1-100; default 10). "
"Use 'amazon' to search for products, 'reddit' for community discussions."
)
args_schema: Type[BaseModel] = _SearchInput
args_schema: type[BaseModel] = _SearchInput

decodo_api_token: SecretStr = Field(
default=SecretStr(""),
Expand Down Expand Up @@ -388,7 +396,7 @@ def _run(
query: str,
engine: str = "google",
num_results: int = 10,
run_manager: Optional[CallbackManagerForToolRun] = None,
run_manager: CallbackManagerForToolRun | None = None,
) -> str:
"""Execute a search and return results as a JSON string.

Expand Down Expand Up @@ -417,21 +425,26 @@ def _run(
target = _ENGINE_TARGET_MAP.get(engine, "google_search")

# Prepend Reddit site filter when using the reddit pseudo-engine.
effective_query = (
f"{_REDDIT_SITE_FILTER} {query}" if engine == "reddit" else query
)
effective_query = f"{_REDDIT_SITE_FILTER} {query}" if engine == "reddit" else query

payload: dict[str, Any] = {
"target": target,
"query": effective_query,
"limit": num_results,
"parse": True,
}

response = _do_scrape(
token, _DEFAULT_BASE_URL, payload, auth_mode=self.auth_mode
)
response = _do_scrape(token, _DEFAULT_BASE_URL, payload, auth_mode=self.auth_mode)
results = response.get("results", [])

# A failed scrape comes back as HTTP 200, either with a failed status
# inside `results` or with no `results` at all (e.g. status 613). Raise
# so agents can retry instead of reading it as "no results".
if not results or all(_is_failed(entry) for entry in results):
status = results[0].get("status_code") if results else response.get("status_code")
message = response.get("message") or "no results returned"
raise RuntimeError(f"Decodo search failed (status {status}): {message}")

serialisable = []
for entry in results:
content = entry.get("content", "")
Expand Down
7 changes: 3 additions & 4 deletions libs/partners/decodo/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "hatchling.build"

[project]
name = "langchain-decodo"
version = "0.1.0"
version = "0.1.1"
description = "LangChain integration for the Decodo web scraping API"
readme = "README.md"
license = { text = "MIT" }
Expand Down Expand Up @@ -41,8 +41,8 @@ test_integration = []

[project.urls]
Homepage = "https://decodo.com"
Documentation = "https://developers.decodo.com"
Repository = "https://github.com/langchain-ai/langchain/tree/master/libs/partners/decodo"
Documentation = "https://help.decodo.com"
Repository = "https://github.com/Decodo/langchain/tree/master/libs/partners/decodo"

[tool.hatch.build.targets.wheel]
packages = ["langchain_decodo"]
Expand All @@ -55,7 +55,6 @@ target-version = "py310"
select = ["E", "F", "I", "UP"]

[tool.mypy]
python_version = "3.10"
strict = true

[tool.pytest.ini_options]
Expand Down
6 changes: 6 additions & 0 deletions libs/partners/decodo/tests/integration_tests/test_compile.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
import pytest


@pytest.mark.compile
def test_placeholder() -> None:
"""Allows CI to verify that the integration tests import and compile."""
Loading
Loading