From ba1e47983ffb4ef420a562644d83e564147429b4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=8F=AD=E6=89=AC?= Date: Tue, 22 Sep 2026 22:28:29 +0800 Subject: [PATCH] enhanced skills --- AGENTS.md | 45 ++ pyproject.toml | 1 + .../engine/context/context_compressor.py | 2 +- src/leapflow/engine/tool_dispatch_engine.py | 6 +- src/leapflow/hub/__init__.py | 30 +- src/leapflow/hub/backends/huggingface.py | 296 ++++++++++++- src/leapflow/hub/client.py | 74 +++- src/leapflow/hub/contribute.py | 295 +++++++++++++ src/leapflow/hub/federated.py | 269 ++++++++++++ src/leapflow/hub/marketplace.py | 353 +++++++++++++++ .../plugins/tool_plugins/skill_discovery.py | 34 +- src/leapflow/skills/__init__.py | 7 + src/leapflow/skills/action_policy.py | 2 +- .../skills/builtin_skills/.skills_index.json | 1 + .../builtin_skills/api_designer/SKILL.md | 189 ++++++++ .../builtin_skills/code_review/SKILL.md | 168 ++++++++ .../builtin_skills/data_analysis/SKILL.md | 193 +++++++++ .../builtin_skills/devops_helper/SKILL.md | 177 ++++++++ .../builtin_skills/document_writer/SKILL.md | 170 ++++++++ .../builtin_skills/git_workflow/SKILL.md | 171 ++++++++ .../builtin_skills/i18n_helper/SKILL.md | 205 +++++++++ .../builtin_skills/project_analysis/SKILL.md | 181 ++++++++ .../builtin_skills/security_audit/SKILL.md | 200 +++++++++ .../builtin_skills/shell_automation/SKILL.md | 166 +++++++ .../builtin_skills/test_writer/SKILL.md | 195 +++++++++ .../builtin_skills/web_research/SKILL.md | 141 ++++++ src/leapflow/skills/curator.py | 211 ++++++++- src/leapflow/skills/discovery.py | 117 ++++- src/leapflow/skills/index.py | 45 +- src/leapflow/skills/injector.py | 20 +- src/leapflow/skills/mcp_bridge.py | 360 ++++++++++++++++ src/leapflow/skills/tool_call_parser.py | 137 ++++++ src/leapflow/skills/tool_executor.py | 258 ++++------- src/leapflow/skills/tool_types.py | 72 ++++ src/leapflow/tools/hub_tool.py | 404 +++++++++++++++++- tests/test_context_disclosure.py | 2 +- tests/test_safety_and_policy.py | 2 +- tests/test_semantic_schema.py | 7 +- tests/test_slash_command_router.py | 2 +- 39 files changed, 4959 insertions(+), 249 deletions(-) create mode 100644 src/leapflow/hub/contribute.py create mode 100644 src/leapflow/hub/federated.py create mode 100644 src/leapflow/hub/marketplace.py create mode 100644 src/leapflow/skills/builtin_skills/.skills_index.json create mode 100644 src/leapflow/skills/builtin_skills/api_designer/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/code_review/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/data_analysis/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/devops_helper/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/document_writer/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/git_workflow/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/i18n_helper/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/project_analysis/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/security_audit/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/shell_automation/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/test_writer/SKILL.md create mode 100644 src/leapflow/skills/builtin_skills/web_research/SKILL.md create mode 100644 src/leapflow/skills/mcp_bridge.py create mode 100644 src/leapflow/skills/tool_call_parser.py create mode 100644 src/leapflow/skills/tool_types.py diff --git a/AGENTS.md b/AGENTS.md index ff77262b..7368132f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -125,6 +125,51 @@ The plugin subsystem is not a feature area — it is how the product is composed - **`self_describe` is the canonical tool for agent self-cognition**: all facets read live runtime state (registry, daemon, engine, build_info) through `bind_runtime` injected services, never from documentation or static config. A new runtime observable (e.g., a new daemon metric, a new engine state) that the agent should be aware of must be wired into the appropriate `self_describe` facet in the same change — an observable that exists only in daemon status but not in any tool is invisible to the agent. - **The plugin contract is published, so it changes with the code**: `docs/plugins/third_party_plugin_development.md` (interfaces, deployment, security model) and `docs/plugins/plugin_lifecycle_management.md` (lifecycle, governance matrix, enforcement status) are third-party-facing specifications whose tables state what the code does *today*. A change to a Protocol, a lifecycle transition, an approval rule, a config key, or an injectable dependency name updates them in the same change — and never promotes a roadmap entry to ENFORCED ahead of the wiring. +## Extension Ladder + +LeapFlow exposes four levels of extension, ordered from lowest barrier to deepest integration. Pick the lowest level that satisfies the requirement — it will ship faster, carry less maintenance cost, and stay compatible across upgrades. + +### Level 1: Skill (`SKILL.md`) — Lowest Barrier + +- Pure Markdown file with YAML frontmatter; no code changes required. +- Add a `SKILL.md` to the skills directory and the agent discovers it at startup. +- Declares tool dependencies, trigger phrases, category, and platform constraints. +- The LLM reads the skill document and autonomously calls existing tools to execute the workflow. +- Compatible with Hermes skill format (`metadata.hermes` namespace). +- **Best for:** custom workflows, domain knowledge, operational playbooks, guided procedures. + +### Level 2: MCP Server — Low Barrier + +- External process communicating via Model Context Protocol (JSON-RPC over stdio/SSE). +- Brings external service capabilities into the agent as discoverable tools. +- Language-agnostic — any runtime that speaks MCP can serve tools. +- **Best for:** external API integrations, third-party service connectors, language-specific tooling. + +### Level 3: Plugin (Python Module) — Medium Barrier + +- Python module implementing the `ToolPlugin` Protocol (`runtime_checkable`). +- Full access to LeapFlow's runtime: EventBus, memory, storage, settings via `bind_runtime`. +- Subject to Progressive Trust lifecycle: DRAFT → CANDIDATE → VERIFIED → PRODUCTION. +- Sandbox isolation via subprocess JSON-RPC until trust is earned. +- **Best for:** deep framework integration, new LLM providers, custom storage backends, platform adapters. + +### Level 4: Core Tool — High Barrier + +- Direct modification to LeapFlow's core tool system (`leapflow/tools/`). +- Requires understanding of internal architecture, review process, and compliance with all rules in this document. +- **Best for:** fundamental capabilities that all plugins and skills may depend on. + +### Summary + +| Level | Mechanism | Barrier | Use Case | Example | +|-------|-----------|---------|----------|---------| +| 1 | Skill (`SKILL.md`) | Lowest | Workflows, playbooks, domain knowledge | Deployment checklist, code-review guide | +| 2 | MCP Server | Low | External services, cross-language tools | GitHub API connector, database explorer | +| 3 | Plugin (Python) | Medium | Runtime integration, providers, adapters | LLM provider, gateway adapter | +| 4 | Core Tool | High | Foundational agent capabilities | File I/O, code search | + +> **Community contributions should start at Level 1 (Skill).** It requires no code changes, has the fastest feedback loop, and can be shared as a single Markdown file. Escalate to a higher level only when the skill layer cannot express the needed capability. + ## Path Tree, Configuration, and Secrets Rules - **Path tree is a product contract**: every LeapFlow-managed path must be declared by `PathLayout`, `ProfileLayout`, `CacheLayout`, or a child layout object. Runtime code must consume layout APIs, never assemble managed paths with ad-hoc string joins. diff --git a/pyproject.toml b/pyproject.toml index 3ba30c08..8637a8c6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -53,6 +53,7 @@ dev = [ "pytest-cov>=5.0", ] hub = ["modelscope-hub>=0.4.5"] +huggingface = ["huggingface-hub>=0.20.0"] # Native Anthropic Messages API provider. Optional: the core install uses the # OpenAI-compatible transport by default; this extra enables AnthropicChat for # endpoints that speak the Anthropic wire format (api.anthropic.com, DeepSeek diff --git a/src/leapflow/engine/context/context_compressor.py b/src/leapflow/engine/context/context_compressor.py index f0ff7cd5..b7c83805 100644 --- a/src/leapflow/engine/context/context_compressor.py +++ b/src/leapflow/engine/context/context_compressor.py @@ -173,7 +173,7 @@ class CompressorConfig: archive_fn: Optional[ArchiveFn] = field(default=None, repr=False) token_count_fn: Optional[TokenCountFn] = field(default=None, repr=False) - # Legacy field aliases (backward compat with engine.py / tool_executor.py) + # Legacy field aliases (backward compat with engine.py / tool_executor.py [deprecated]) threshold: int = 16 keep_tail: int = 4 max_output_chars: int = 2000 diff --git a/src/leapflow/engine/tool_dispatch_engine.py b/src/leapflow/engine/tool_dispatch_engine.py index 498f2ec3..4b9c2ba1 100644 --- a/src/leapflow/engine/tool_dispatch_engine.py +++ b/src/leapflow/engine/tool_dispatch_engine.py @@ -368,11 +368,11 @@ def _format_tool_catalog(tool_definitions: List[Dict[str, Any]]) -> str: def _parse_tool_call_from_content(content: str) -> Optional[Dict[str, Any]]: """Extract tool call from LLM response content. - Reuses the robust parser from tool_executor. + Reuses the robust parser from tool_call_parser. """ - from leapflow.skills.tool_executor import _parse_tool_call + from leapflow.skills.tool_call_parser import parse_tool_call - call = _parse_tool_call(content) + call = parse_tool_call(content) if call: return {"name": call.name, "arguments": call.params} return None diff --git a/src/leapflow/hub/__init__.py b/src/leapflow/hub/__init__.py index 236cc674..d43bdf31 100644 --- a/src/leapflow/hub/__init__.py +++ b/src/leapflow/hub/__init__.py @@ -2,13 +2,15 @@ """LeapFlow Hub — cloud collaboration for skill sharing and multi-device sync. Public API: - HubClient - Main facade for push/pull/search/sync - SyncEngine - Bidirectional sync engine with conflict resolution - SyncAction - Single sync operation descriptor - SyncPlan - Computed synchronization plan (from sync.py) - SkillSerializer - Bundle serialization/deserialization - ContentSanitizer - Pre-push content scanning - SecurityAuditor - Post-pull code auditing + HubClient - Main facade for push/pull/search/sync + FederatedHubRouter - Multi-backend parallel search aggregator + FederatedSearchResult - Aggregated search outcome container + SyncEngine - Bidirectional sync engine with conflict resolution + SyncAction - Single sync operation descriptor + SyncPlan - Computed synchronization plan (from sync.py) + SkillSerializer - Bundle serialization/deserialization + ContentSanitizer - Pre-push content scanning + SecurityAuditor - Post-pull code auditing Protocol & Types (from protocol.py): HubBackend, SkillBundle, SkillManifest, SkillSummary, @@ -28,17 +30,31 @@ ) from leapflow.hub.client import HubClient +from leapflow.hub.federated import FederatedHubRouter, FederatedSearchResult from leapflow.hub.security import ContentSanitizer, SanitizationWarning, SecurityAuditor from leapflow.hub.serializer import SkillSerializer from leapflow.hub.sync import SyncAction, SyncEngine, SyncPlan +from leapflow.hub.marketplace import MarketplaceCategory, MarketplaceEntry, SkillMarketplace +from leapflow.hub.contribute import CommunityContributor, ContributionRecord, ContributionStatus __all__ = [ # Client "HubClient", + # Federated + "FederatedHubRouter", + "FederatedSearchResult", # Sync "SyncEngine", "SyncPlan", "SyncAction", + # Marketplace + "MarketplaceCategory", + "MarketplaceEntry", + "SkillMarketplace", + # Community contribution + "CommunityContributor", + "ContributionRecord", + "ContributionStatus", # Serialization "SkillSerializer", # Security diff --git a/src/leapflow/hub/backends/huggingface.py b/src/leapflow/hub/backends/huggingface.py index c695493a..642938a4 100644 --- a/src/leapflow/hub/backends/huggingface.py +++ b/src/leapflow/hub/backends/huggingface.py @@ -1,13 +1,25 @@ # Copyright (c) Alibaba, Inc. and its affiliates. -"""HuggingFace Hub backend — placeholder for Phase 2. +"""HuggingFace Hub backend — push/pull/search skills via huggingface_hub SDK. -Will be activated when huggingface-hub SDK integration is ready. +Implements HubBackend Protocol using the ``huggingface_hub`` library (HfApi). +Stores skills as HuggingFace datasets (repo_type="dataset") since skill +bundles are not ML models. + +Authentication: ``HF_TOKEN`` or ``HUGGINGFACE_TOKEN`` env var, or a prior +``huggingface-cli login`` session. + +All synchronous HfApi calls are wrapped with ``asyncio.to_thread`` to avoid +blocking the event loop (same pattern as the ModelScope backend). """ from __future__ import annotations +import asyncio import logging -from typing import List, Optional +import os +import tempfile +from pathlib import Path +from typing import Any, List, Optional from leapflow.hub.protocol import ( PushResult, @@ -17,27 +29,109 @@ VersionInfo, Visibility, ) +from leapflow.hub.serializer import SkillSerializer logger = logging.getLogger(__name__) -_NOT_IMPLEMENTED_MSG = ( - "HuggingFace backend is not yet implemented. " - "This integration will be available in a future release." +_SDK_INSTALL_HINT = ( + "HuggingFace Hub SDK not found. Install it with:\n" + " pip install huggingface-hub\n" + "or:\n" + " uv pip install huggingface-hub" ) +_TOKEN_MISSING_MSG = ( + "HuggingFace token not found. Set HF_TOKEN or HUGGINGFACE_TOKEN " + "environment variable, or run 'huggingface-cli login'." +) + +# Map LeapFlow visibility levels to HuggingFace visibility. +_VISIBILITY_MAP = { + Visibility.PRIVATE: "private", + Visibility.INTERNAL: "private", # HF has no "internal"; fall back to private + Visibility.PUBLIC: "public", +} + +# Default timeout for HfApi network calls (seconds). +_DEFAULT_TIMEOUT = 60 + + +def _resolve_token() -> str: + """Resolve HuggingFace token from environment variables. + + Checks ``HF_TOKEN`` first (the canonical variable used by ``huggingface_hub`` + itself), then falls back to ``HUGGINGFACE_TOKEN``. + """ + return ( + os.environ.get("HF_TOKEN", "").strip() + or os.environ.get("HUGGINGFACE_TOKEN", "").strip() + ) + class HuggingFaceBackend: - """HubBackend implementation for HuggingFace Hub. + """HubBackend implementation for HuggingFace Hub (huggingface.co). - Not yet implemented. Raises NotImplementedError for all operations. - Will be activated when huggingface-hub SDK integration is ready. + Wraps the ``huggingface_hub.HfApi`` SDK with an async interface and + friendly error handling. Skills are stored as HuggingFace **datasets** + (``repo_type="dataset"``) rather than models, since skill bundles are + source artefacts, not trained weights. """ hub_type = "huggingface" + def __init__(self, token: str = "") -> None: + """Initialize HuggingFace backend. + + Args: + token: Explicit HF token. If empty, resolved from env or + cached CLI login session. + + Raises: + ImportError: If ``huggingface_hub`` is not installed. + """ + self._token: str = token or _resolve_token() + self._api: Any = None + self._serializer = SkillSerializer() + self._ensure_sdk() + + # ── SDK bootstrap ──────────────────────────────────────────────────── + + def _ensure_sdk(self) -> None: + """Verify SDK availability and create API instance.""" + try: + from huggingface_hub import HfApi # type: ignore[import-untyped] + + # Pass token only when explicitly available; HfApi also reads + # HF_TOKEN / cached login automatically. + kwargs: dict[str, Any] = {} + if self._token: + kwargs["token"] = self._token + self._api = HfApi(**kwargs) + except ImportError: + raise ImportError(_SDK_INSTALL_HINT) from None + + # ── HubBackend Protocol ────────────────────────────────────────────── + async def authenticate(self) -> UserInfo: - """Not implemented.""" - raise NotImplementedError(_NOT_IMPLEMENTED_MSG) + """Authenticate with HuggingFace and return current user info.""" + try: + info = await asyncio.to_thread(self._api.whoami) + return UserInfo( + username=info.get("name", info.get("fullname", "")), + email=info.get("email", ""), + avatar_url=info.get("avatarUrl", ""), + ) + except Exception as e: + error_msg = str(e).lower() + if "unauthorized" in error_msg or "401" in error_msg: + raise RuntimeError( + f"HuggingFace authentication failed: {e}. {_TOKEN_MISSING_MSG}" + ) from e + raise RuntimeError( + f"HuggingFace authentication failed: {e}. " + "Ensure you have logged in via 'huggingface-cli login' or " + "set the HF_TOKEN environment variable." + ) from e async def push_skill( self, @@ -45,29 +139,189 @@ async def push_skill( repo_id: str, visibility: Visibility = Visibility.PRIVATE, ) -> PushResult: - """Not implemented.""" - raise NotImplementedError(_NOT_IMPLEMENTED_MSG) + """Push a skill bundle to HuggingFace Hub. + + Creates the dataset repository if it doesn't exist, then uploads + all bundle files in a single commit. + """ + # Ensure repository exists + await self._ensure_repo(repo_id, visibility) + + # Serialize bundle to files + files = self._serializer.bundle_to_files(bundle) + version = bundle.manifest.version or "0.1.0" + commit_message = f"Push skill {bundle.manifest.name} v{version}" + + with tempfile.TemporaryDirectory(prefix="leapflow_hf_push_") as tmp_dir: + tmp_path = Path(tmp_dir) + for filename, content in files.items(): + file_path = tmp_path / filename + file_path.write_text(content, encoding="utf-8") + + # Upload the entire folder as a single commit + await asyncio.to_thread( + self._api.upload_folder, + repo_id=repo_id, + folder_path=str(tmp_path), + repo_type="dataset", + commit_message=commit_message, + ) + + # Construct result URL + url = f"https://huggingface.co/datasets/{repo_id}" + + logger.info( + "Pushed skill '%s' v%s to %s", bundle.manifest.name, version, url + ) + + return PushResult( + repo_id=repo_id, + version=version, + url=url, + hub_type=self.hub_type, + ) async def pull_skill( self, repo_id: str, version: Optional[str] = None, ) -> SkillBundle: - """Not implemented.""" - raise NotImplementedError(_NOT_IMPLEMENTED_MSG) + """Pull a skill bundle from HuggingFace Hub.""" + try: + kwargs: dict[str, Any] = { + "repo_id": repo_id, + "repo_type": "dataset", + } + if version: + kwargs["revision"] = version + + local_dir: str = await asyncio.to_thread( + self._api.snapshot_download, + **kwargs, + ) + + # Read all files in a thread to avoid blocking the event loop + def _read_files(local_path: Path) -> dict[str, str]: + files: dict[str, str] = {} + for file_path in local_path.rglob("*"): + if file_path.is_file() and not file_path.name.startswith("."): + rel = file_path.relative_to(local_path) + files[str(rel)] = file_path.read_text(encoding="utf-8") + return files + + local_path = Path(local_dir) + files = await asyncio.to_thread(_read_files, local_path) + + return self._serializer.files_to_bundle(files) + + except Exception as e: + raise RuntimeError( + f"Failed to pull skill '{repo_id}' from HuggingFace: {e}" + ) from e async def list_remote_skills( self, owner: Optional[str] = None, query: Optional[str] = None, ) -> List[SkillSummary]: - """Not implemented.""" - raise NotImplementedError(_NOT_IMPLEMENTED_MSG) + """List skills available on HuggingFace Hub. + + Searches datasets with an optional author filter and text query. + """ + try: + kwargs: dict[str, Any] = {} + if owner: + kwargs["author"] = owner + if query: + kwargs["search"] = query + + datasets = await asyncio.to_thread( + self._api.list_datasets, **kwargs + ) + + skills: List[SkillSummary] = [] + for ds in datasets: + repo_id = getattr(ds, "id", "") + name = repo_id.split("/")[-1] if "/" in repo_id else repo_id + skills.append( + SkillSummary( + repo_id=repo_id, + name=name, + description=getattr(ds, "description", "") or "", + version=getattr(ds, "sha", ""), + downloads=getattr(ds, "downloads", 0) or 0, + hub_type=self.hub_type, + ) + ) + return skills + + except Exception as e: + logger.warning("Failed to list skills from HuggingFace: %s", e) + return [] async def get_skill_versions(self, repo_id: str) -> List[VersionInfo]: - """Not implemented.""" - raise NotImplementedError(_NOT_IMPLEMENTED_MSG) + """Get version history for a skill on HuggingFace. + + Uses commit history on the dataset repository as version records. + """ + try: + commits = await asyncio.to_thread( + self._api.list_repo_commits, + repo_id=repo_id, + repo_type="dataset", + ) + + versions: List[VersionInfo] = [] + for commit in commits: + versions.append( + VersionInfo( + version=getattr(commit, "title", "") or "", + created_at=str(getattr(commit, "created_at", "") or ""), + commit_sha=getattr(commit, "commit_id", "") or "", + ) + ) + return versions + + except Exception as e: + logger.warning( + "Failed to get versions for '%s' on HuggingFace: %s", + repo_id, + e, + ) + return [] async def delete_skill(self, repo_id: str) -> None: - """Not implemented.""" - raise NotImplementedError(_NOT_IMPLEMENTED_MSG) + """Delete a skill dataset repository from HuggingFace Hub.""" + try: + await asyncio.to_thread( + self._api.delete_repo, + repo_id=repo_id, + repo_type="dataset", + ) + logger.info("Deleted HuggingFace dataset repository: %s", repo_id) + except Exception as e: + raise RuntimeError( + f"Failed to delete skill '{repo_id}' from HuggingFace: {e}" + ) from e + + # ─── Private Helpers ───────────────────────────────────────────────── + + async def _ensure_repo(self, repo_id: str, visibility: Visibility) -> None: + """Create dataset repository if it doesn't exist.""" + hf_visibility = _VISIBILITY_MAP.get(visibility, "private") + try: + await asyncio.to_thread( + self._api.create_repo, + repo_id=repo_id, + repo_type="dataset", + private=(hf_visibility == "private"), + exist_ok=True, + ) + except Exception as e: + error_str = str(e).lower() + if "already exists" in error_str or "exist" in error_str: + logger.debug("HuggingFace dataset %s already exists", repo_id) + else: + raise RuntimeError( + f"Failed to create HuggingFace dataset '{repo_id}': {e}" + ) from e diff --git a/src/leapflow/hub/client.py b/src/leapflow/hub/client.py index 1bb6750c..b5fce802 100644 --- a/src/leapflow/hub/client.py +++ b/src/leapflow/hub/client.py @@ -9,17 +9,20 @@ import asyncio import logging +from dataclasses import replace from typing import Callable, Dict, List, Optional, Tuple, TYPE_CHECKING if TYPE_CHECKING: # hub.sync imports HubClient, so this stays annotation-only to avoid a cycle. from leapflow.hub.sync import SyncPlan + from leapflow.hub.federated import FederatedHubRouter, FederatedSearchResult from leapflow.hub.protocol import ( HubBackend, PushResult, SkillBundle, SkillManifest, + SkillSourceTag, SkillSummary, UserInfo, VersionConflictError, @@ -116,6 +119,7 @@ def __init__( ] self._backend: Optional[HubBackend] = None self._backend_cache: Dict[str, HubBackend] = {} + self._federated: Optional[FederatedHubRouter] = None @property def hub_type(self) -> str: @@ -263,17 +267,37 @@ async def pull( """Pull a skill bundle from the remote hub. Supports identifier routing: github://owner/repo, hf://owner/repo, etc. + Pulled bundles are automatically tagged for Progressive Trust: + source_tag is set to HUB and tier is clamped to DRAFT (1). Args: repo_id: Repository identifier (may include protocol prefix). version: Specific version (None = latest). Returns: - Complete SkillBundle. + Complete SkillBundle with trust metadata applied. """ backend_name, normalized = self._route_identifier(repo_id) backend = self._get_backend_for(backend_name) - return await backend.pull_skill(normalized, version) + bundle = await backend.pull_skill(normalized, version) + return self._apply_hub_trust(bundle, backend_name) + + @staticmethod + def _apply_hub_trust(bundle: SkillBundle, backend_name: str) -> SkillBundle: + """Ensure a pulled bundle carries Hub Progressive Trust metadata. + + Sets source_tag to 'hub' and clamps tier to DRAFT (1) so the skill + enters the trust pipeline at the lowest level. + """ + manifest = bundle.manifest + new_tier = min(manifest.tier, 1) # DRAFT is the ceiling for hub pulls + updated_manifest = replace( + manifest, + source_tag=SkillSourceTag.HUB.value, + tier=new_tier, + hub_type=backend_name, + ) + return replace(bundle, manifest=updated_manifest) async def search( self, @@ -376,6 +400,50 @@ async def sync_skills( return ClientSyncPlan(to_push=to_push, to_pull=to_pull, conflicts=conflicts) + # ─── Federated Search ───────────────────────────────────────────────── + + def _get_or_create_federated(self) -> FederatedHubRouter: + """Lazily build a FederatedHubRouter populated with all registered backends.""" + if self._federated is not None: + return self._federated + + from leapflow.hub.federated import FederatedHubRouter + + router = FederatedHubRouter() + for hub_type in self._search_sources: + try: + backend = self._get_backend_for(hub_type) + router.add_backend(hub_type, backend) + except ValueError: + logger.warning( + "Federated search: backend '%s' not available — skipped", + hub_type, + ) + self._federated = router + return router + + async def federated_search( + self, + query: str, + owner: str | None = None, + ) -> FederatedSearchResult: + """Search across ALL registered backends via FederatedHubRouter. + + Creates or reuses a FederatedHubRouter populated with every backend + listed in ``search_sources``. Results are deduplicated and sorted. + + Args: + query: Free-text search query. + owner: Optional owner/org filter. + + Returns: + FederatedSearchResult with merged, deduplicated summaries. + """ + router = self._get_or_create_federated() + return await router.search(query, owner=owner) + + # ─── Authentication ─────────────────────────────────────────────────── + async def login(self) -> UserInfo: """Authenticate with the hub backend. @@ -415,7 +483,7 @@ def _modelscope_factory() -> HubBackend: HubClient.register("modelscope", _modelscope_factory) - # HuggingFace backend — placeholder + # HuggingFace backend — available if huggingface-hub installed def _huggingface_factory() -> HubBackend: from leapflow.hub.backends.huggingface import HuggingFaceBackend diff --git a/src/leapflow/hub/contribute.py b/src/leapflow/hub/contribute.py new file mode 100644 index 00000000..fa6c0e28 --- /dev/null +++ b/src/leapflow/hub/contribute.py @@ -0,0 +1,295 @@ +# Copyright (c) Alibaba, Inc. and its affiliates. +"""Community contribution pipeline — prepare, submit, and track skill contributions. + +Provides a structured workflow for contributing skills back to the Hub: +DRAFT → SUBMITTED → UNDER_REVIEW → APPROVED/REJECTED → PUBLISHED. +Contributions are tracked locally in a JSON file per profile. +""" + +from __future__ import annotations + +import enum +import json +import logging +import time +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any, Dict, List, Optional + +logger = logging.getLogger(__name__) + + +# ─── Data Types ────────────────────────────────────────────────────────────── + + +@enum.unique +class ContributionStatus(enum.Enum): + """Lifecycle status of a community skill contribution.""" + + DRAFT = "draft" + SUBMITTED = "submitted" + UNDER_REVIEW = "under_review" + APPROVED = "approved" + REJECTED = "rejected" + PUBLISHED = "published" + + +@dataclass(frozen=True) +class ContributionRecord: + """Persistent record tracking one skill contribution.""" + + skill_name: str + author: str + status: str # ContributionStatus value + submitted_at: str = "" + review_notes: str = "" + hub_type: str = "" + repo_id: str = "" + + +# ─── Contributor Pipeline ──────────────────────────────────────────────────── + + +class CommunityContributor: + """End-to-end community contribution pipeline. + + Validates, sanitizes, and submits local skills to the Hub, + then tracks the contribution lifecycle in a local JSON store. + + Args: + hub_client: HubClient instance for push operations. + store_path: Path to the contributions JSON file. + """ + + def __init__( + self, + hub_client: Any, + *, + store_path: Optional[Path] = None, + ) -> None: + self._hub = hub_client + self._store_path = store_path + self._records: Dict[str, ContributionRecord] = {} + self._loaded = False + + # ── Public API ──────────────────────────────────────────────────────── + + async def prepare(self, skill_name: str, *, ctx: Any = None) -> str: + """Validate a local skill and generate a sanitized SkillBundle. + + Runs ContentSanitizer to detect secrets/PII before submission. + + Args: + skill_name: Name of the local skill to prepare. + ctx: Optional runtime context with skill_lib access. + + Returns: + Human-readable preparation report. + """ + from leapflow.hub.security import ContentSanitizer + from leapflow.hub.serializer import SkillSerializer + + # Load skill data from context + stored_dict = self._load_skill_from_ctx(skill_name, ctx) + if isinstance(stored_dict, str): + return stored_dict # error message + + # Serialize to bundle + serializer = SkillSerializer() + bundle = serializer.export_skill(stored_dict) + + # Sanitize + sanitizer = ContentSanitizer() + warnings = sanitizer.scan(bundle) + + high = sum(1 for w in warnings if w.severity == "high") + medium = sum(1 for w in warnings if w.severity == "medium") + + # Create or update draft record + self._ensure_loaded() + record = ContributionRecord( + skill_name=skill_name, + author=stored_dict.get("author", ""), + status=ContributionStatus.DRAFT.value, + ) + self._records[skill_name] = record + self._save_store() + + lines = [f"Prepared '{skill_name}' for contribution."] + if warnings: + lines.append(f" Sanitization: {high} high, {medium} medium, " + f"{len(warnings) - high - medium} low warning(s).") + if high > 0: + lines.append(" ⚠ High-risk issues must be resolved before submission.") + for w in warnings: + if w.severity == "high": + lines.append(f" - {w.detail}") + else: + lines.append(" ✓ No sanitization warnings — ready to submit.") + lines.append(f" Status: {ContributionStatus.DRAFT.value}") + return "\n".join(lines) + + async def submit( + self, + skill_name: str, + hub_type: str = "github", + *, + ctx: Any = None, + ) -> str: + """Push skill to hub with PUBLIC visibility and create contribution record. + + Args: + skill_name: Name of the local skill to submit. + hub_type: Target hub backend (default: 'github'). + ctx: Optional runtime context with skill_lib access. + + Returns: + Human-readable submission result. + """ + from leapflow.hub.protocol import Visibility + from leapflow.hub.security import ContentSanitizer + from leapflow.hub.serializer import SkillSerializer + + stored_dict = self._load_skill_from_ctx(skill_name, ctx) + if isinstance(stored_dict, str): + return stored_dict + + serializer = SkillSerializer() + bundle = serializer.export_skill(stored_dict) + + # Pre-flight sanitization check + sanitizer = ContentSanitizer() + warnings = sanitizer.scan(bundle) + high_warnings = [w for w in warnings if w.severity == "high"] + if high_warnings: + return ( + f"Cannot submit '{skill_name}': {len(high_warnings)} high-risk " + f"sanitization warning(s). Run prepare() first to review." + ) + + # Push to hub + try: + result = await self._hub.push( + bundle, + skill_name=skill_name, + visibility=Visibility.PUBLIC, + ) + except Exception as exc: + return f"Submission failed: {type(exc).__name__}: {exc}" + + # Record contribution + self._ensure_loaded() + actual_hub_type = getattr(self._hub, 'hub_type', hub_type) + record = ContributionRecord( + skill_name=skill_name, + author=stored_dict.get("author", ""), + status=ContributionStatus.SUBMITTED.value, + submitted_at=time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + hub_type=actual_hub_type, + repo_id=result.repo_id, + ) + self._records[skill_name] = record + self._save_store() + + return ( + f"Submitted '{skill_name}' to {result.repo_id} ({actual_hub_type}).\n" + f" Version: v{result.version}\n" + f" URL: {result.url}\n" + f" Status: {ContributionStatus.SUBMITTED.value}" + ) + + def check_status(self, skill_name: str) -> ContributionStatus: + """Return the current ContributionStatus for a skill. + + Args: + skill_name: Name of the contributed skill. + + Returns: + Current status enum value. + + Raises: + KeyError: If no contribution record exists for the skill. + """ + self._ensure_loaded() + record = self._records.get(skill_name) + if record is None: + raise KeyError(f"No contribution record for '{skill_name}'.") + return ContributionStatus(record.status) + + def list_my_contributions(self) -> List[ContributionRecord]: + """Return all contribution records from local store.""" + self._ensure_loaded() + return list(self._records.values()) + + # ── Private Helpers ─────────────────────────────────────────────────── + + @staticmethod + def _load_skill_from_ctx( + skill_name: str, ctx: Any + ) -> Dict[str, Any] | str: + """Load skill data from context's skill library. + + Returns dict on success, or an error string on failure. + """ + if ctx is None or not hasattr(ctx, "skill_lib") or ctx.skill_lib is None: + return "Error: Skill library context not available." + + try: + stored = ctx.skill_lib.load_skill_by_title(skill_name) + if stored is None: + return f"Error: Skill '{skill_name}' not found in local library." + return { + "name": getattr(stored, "title", skill_name), + "version": getattr(stored, "version", "0.1.0"), + "description": getattr(stored, "description", ""), + "source_code": getattr(stored, "source_code", ""), + "parameters": getattr(stored, "parameters", []), + "triggers": list(getattr(stored, "trigger_phrases", [])), + "trajectory_skeleton": getattr(stored, "trajectory_skeleton", ""), + "copilot_prior": getattr(stored, "copilot_prior", ""), + "readme": getattr(stored, "readme", f"# {skill_name}\n"), + "source_tag": getattr(stored, "source_tag", "learned"), + "tier": getattr(stored, "tier", 1), + "author": getattr(stored, "author", ""), + } + except Exception as exc: + return f"Error loading skill '{skill_name}': {exc}" + + def _ensure_loaded(self) -> None: + if self._loaded: + return + self._load_store() + self._loaded = True + + def _load_store(self) -> None: + """Read contributions.json from disk.""" + if self._store_path is None or not self._store_path.exists(): + return + try: + raw = json.loads(self._store_path.read_text(encoding="utf-8")) + for item in raw.get("contributions", []): + try: + record = ContributionRecord(**item) + self._records[record.skill_name] = record + except (TypeError, AttributeError): + logger.debug("Skipping malformed contribution entry: %r", item) + except (json.JSONDecodeError, OSError) as exc: + logger.warning("Failed to load contributions store: %s", exc) + + def _save_store(self) -> None: + """Persist contribution records to disk.""" + if self._store_path is None: + return + try: + self._store_path.parent.mkdir(parents=True, exist_ok=True) + data = { + "version": 1, + "updated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + "contributions": [asdict(r) for r in self._records.values()], + } + self._store_path.write_text( + json.dumps(data, indent=2, ensure_ascii=False), + encoding="utf-8", + ) + except OSError as exc: + logger.warning("Failed to save contributions store: %s", exc) diff --git a/src/leapflow/hub/federated.py b/src/leapflow/hub/federated.py new file mode 100644 index 00000000..98281b4a --- /dev/null +++ b/src/leapflow/hub/federated.py @@ -0,0 +1,269 @@ +# Copyright (c) Alibaba, Inc. and its affiliates. +"""Federated Hub Router — parallel multi-backend search and pull aggregation. + +Routes search/list queries to multiple Hub backends concurrently, aggregates +and deduplicates results. This is a composition layer above HubBackend; it +does not modify the Protocol itself. +""" + +from __future__ import annotations + +import asyncio +import logging +from dataclasses import dataclass +from typing import Dict, List, Optional, Tuple + +from leapflow.hub.protocol import HubBackend, SkillBundle, SkillSummary + +logger = logging.getLogger(__name__) + +# Per-backend query timeout (seconds). +_DEFAULT_BACKEND_TIMEOUT: float = 30.0 + + +# ─── Result Container ──────────────────────────────────────────────────────── + + +@dataclass(frozen=True) +class FederatedSearchResult: + """Aggregated search outcome across multiple backends.""" + + results: Tuple[SkillSummary, ...] = () + failed_backends: Tuple[str, ...] = () + total_backends_queried: int = 0 + + +# ─── Deduplication Helpers ──────────────────────────────────────────────────── + + +def _semver_tuple(version: str) -> Tuple[int, ...]: + """Parse a version string into a comparable int tuple.""" + cleaned = version.lstrip("vV") + parts: list[int] = [] + for segment in cleaned.split("."): + if segment.isdigit(): + parts.append(int(segment)) + return tuple(parts) if parts else (0,) + + +def _prefer_best(existing: SkillSummary, challenger: SkillSummary) -> SkillSummary: + """Given two summaries for the same skill, return the better one. + + Preference order: + 1. Higher semantic version. + 2. More downloads (popularity signal). + 3. Keep existing (first-seen wins as tie-breaker). + """ + ev = _semver_tuple(existing.version) + cv = _semver_tuple(challenger.version) + if cv > ev: + return challenger + if cv < ev: + return existing + # Versions equal — prefer more downloads. + if challenger.downloads > existing.downloads: + return challenger + return existing + + +def _deduplicate(summaries: List[SkillSummary]) -> List[SkillSummary]: + """Deduplicate by (name, version-agnostic key) keeping the best variant.""" + best_by_name: Dict[str, SkillSummary] = {} + for s in summaries: + key = s.name + if key in best_by_name: + best_by_name[key] = _prefer_best(best_by_name[key], s) + else: + best_by_name[key] = s + return list(best_by_name.values()) + + +# ─── FederatedHubRouter ────────────────────────────────────────────────────── + + +class FederatedHubRouter: + """Routes search/list queries to multiple Hub backends in parallel, + aggregates and deduplicates results. + + Usage:: + + router = FederatedHubRouter() + router.add_backend("modelscope", ms_backend) + router.add_backend("github", gh_backend) + results = await router.search("file management") + """ + + def __init__( + self, + *, + timeout: float = _DEFAULT_BACKEND_TIMEOUT, + ) -> None: + """Initialize the router. + + Args: + timeout: Per-backend query timeout in seconds. + """ + self._backends: List[Tuple[str, HubBackend]] = [] + self._timeout = timeout + + # ─── Registration ───────────────────────────────────────────────────── + + def add_backend(self, hub_type: str, backend: HubBackend) -> None: + """Register a backend for federated queries. + + Args: + hub_type: Backend identifier (e.g. 'modelscope', 'github'). + backend: A concrete HubBackend instance. + """ + # Avoid duplicate registrations for the same hub_type. + for existing_type, _ in self._backends: + if existing_type == hub_type: + logger.debug( + "Backend '%s' already registered — skipping duplicate", hub_type + ) + return + self._backends.append((hub_type, backend)) + logger.debug("Federated router: added backend '%s'", hub_type) + + @property + def backend_count(self) -> int: + """Return the number of registered backends.""" + return len(self._backends) + + @property + def backend_types(self) -> List[str]: + """Return registered backend type names in registration order.""" + return [ht for ht, _ in self._backends] + + # ─── Search ─────────────────────────────────────────────────────────── + + async def search( + self, + query: str, + owner: Optional[str] = None, + ) -> FederatedSearchResult: + """Search all registered backends concurrently and return merged results. + + Failed backends are logged but never block results from healthy ones. + + Args: + query: Free-text search query. + owner: Optional owner/org filter applied to every backend. + + Returns: + FederatedSearchResult with deduplicated summaries. + """ + if not self._backends: + logger.warning("FederatedHubRouter.search called with no backends registered") + return FederatedSearchResult(total_backends_queried=0) + + async def _query_one(hub_type: str, backend: HubBackend) -> List[SkillSummary]: + """Query a single backend with timeout protection.""" + try: + return await asyncio.wait_for( + backend.list_remote_skills(owner=owner, query=query), + timeout=self._timeout, + ) + except asyncio.TimeoutError: + logger.warning( + "Federated search: backend '%s' timed out after %.1fs", + hub_type, + self._timeout, + ) + raise + except Exception: + logger.warning( + "Federated search: backend '%s' failed", + hub_type, + exc_info=True, + ) + raise + + tasks = [ + _query_one(hub_type, backend) for hub_type, backend in self._backends + ] + raw_results = await asyncio.gather(*tasks, return_exceptions=True) + + # Collect successes and failures. + all_summaries: List[SkillSummary] = [] + failed: List[str] = [] + for (hub_type, _backend), result in zip(self._backends, raw_results): + if isinstance(result, BaseException): + failed.append(hub_type) + elif isinstance(result, list): + all_summaries.extend(result) + + deduplicated = _deduplicate(all_summaries) + + # Sort by relevance proxy: downloads desc, then name asc. + deduplicated.sort(key=lambda s: (-s.downloads, s.name)) + + return FederatedSearchResult( + results=tuple(deduplicated), + failed_backends=tuple(failed), + total_backends_queried=len(self._backends), + ) + + # ─── Pull Best Match ────────────────────────────────────────────────── + + async def pull_best( + self, + name_or_query: str, + version: Optional[str] = None, + ) -> Optional[SkillBundle]: + """Search across backends and pull the best matching skill bundle. + + Finds the best match by name across all backends, then pulls from + the backend that owns it. + + Args: + name_or_query: Skill name or search query. + version: Specific version to pull (None = latest). + + Returns: + SkillBundle if found, None otherwise. + """ + search_result = await self.search(name_or_query) + if not search_result.results: + logger.info("Federated pull_best: no results for '%s'", name_or_query) + return None + + best = search_result.results[0] + + # Find the backend that owns this result. + target_backend: Optional[HubBackend] = None + for hub_type, backend in self._backends: + if hub_type == best.hub_type: + target_backend = backend + break + + if target_backend is None: + # Fallback: try the first available backend with the repo_id. + for _hub_type, backend in self._backends: + try: + bundle = await asyncio.wait_for( + backend.pull_skill(best.repo_id, version), + timeout=self._timeout, + ) + return bundle + except Exception: + continue + logger.warning( + "Federated pull_best: could not pull '%s' from any backend", + best.repo_id, + ) + return None + + try: + return await asyncio.wait_for( + target_backend.pull_skill(best.repo_id, version), + timeout=self._timeout, + ) + except Exception: + logger.warning( + "Federated pull_best: pull from '%s' failed for '%s'", + best.hub_type, + best.repo_id, + exc_info=True, + ) + return None diff --git a/src/leapflow/hub/marketplace.py b/src/leapflow/hub/marketplace.py new file mode 100644 index 00000000..f72f6b61 --- /dev/null +++ b/src/leapflow/hub/marketplace.py @@ -0,0 +1,353 @@ +# Copyright (c) Alibaba, Inc. and its affiliates. +"""Skill Marketplace — curated directory with categories, featured/trending skills. + +Provides a browsable marketplace layer on top of HubClient, with offline +support via a local JSON manifest cache. The marketplace composes existing +Hub infrastructure (search, pull) and never replaces it. +""" + +from __future__ import annotations + +import json +import logging +import time +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any, Dict, List, Optional, Sequence, Tuple + +from leapflow.hub.protocol import SkillSummary + +logger = logging.getLogger(__name__) + + +# ─── Data Types ────────────────────────────────────────────────────────────── + + +@dataclass(frozen=True) +class MarketplaceCategory: + """A single browsable category in the skill marketplace.""" + + name: str + description: str + icon: str + skill_count: int = 0 + + +@dataclass(frozen=True) +class MarketplaceEntry: + """Extended skill listing enriched with marketplace metadata. + + Carries all fields from SkillSummary plus curation signals + (featured, trending, rating, install_count, etc.). + """ + + repo_id: str + name: str + description: str = "" + version: str = "" + downloads: int = 0 + hub_type: str = "" + # ── marketplace extensions ── + featured: bool = False + trending: bool = False + rating: float = 0.0 + install_count: int = 0 + updated_at: str = "" + categories: Tuple[str, ...] = () + author: str = "" + + @classmethod + def from_summary( + cls, + summary: SkillSummary, + *, + featured: bool = False, + trending: bool = False, + rating: float = 0.0, + install_count: int = 0, + updated_at: str = "", + categories: Tuple[str, ...] = (), + author: str = "", + ) -> MarketplaceEntry: + """Construct a MarketplaceEntry from an existing SkillSummary.""" + return cls( + repo_id=summary.repo_id, + name=summary.name, + description=summary.description, + version=summary.version, + downloads=summary.downloads, + hub_type=summary.hub_type, + featured=featured, + trending=trending, + rating=rating, + install_count=install_count, + updated_at=updated_at, + categories=categories, + author=author, + ) + + +# ─── Predefined Categories ────────────────────────────────────────────────── + +_DEFAULT_CATEGORIES: Tuple[MarketplaceCategory, ...] = ( + MarketplaceCategory("development", "Software development and coding tools", "🛠️"), + MarketplaceCategory("research", "Academic and scientific research aids", "🔬"), + MarketplaceCategory("automation", "Workflow and task automation", "⚙️"), + MarketplaceCategory("productivity", "Personal and team productivity boosters", "📈"), + MarketplaceCategory("security", "Security scanning and hardening", "🔒"), + MarketplaceCategory("operations", "DevOps, SRE, and infrastructure management", "🖥️"), + MarketplaceCategory("analysis", "Data analysis and visualization", "📊"), + MarketplaceCategory("integration", "Third-party service connectors and bridges", "🔗"), +) + + +# ─── Marketplace ───────────────────────────────────────────────────────────── + + +class SkillMarketplace: + """Curated marketplace layer on top of HubClient. + + Provides category browsing, featured/trending discovery, install + statistics tracking, and offline support via a local JSON cache. + + Args: + hub_client: HubClient instance for hub operations. + cache_path: Path to the local marketplace cache JSON file. + manifest_url: Optional URL to a remote JSON catalog (reserved for + future remote manifest fetching). + """ + + def __init__( + self, + hub_client: Any, + *, + cache_path: Optional[Path] = None, + manifest_url: str = "", + ) -> None: + self._hub = hub_client + self._manifest_url = manifest_url + self._cache_path = cache_path + self._entries: List[MarketplaceEntry] = [] + self._install_stats: Dict[str, int] = {} + self._loaded = False + + # ── Category API ────────────────────────────────────────────────────── + + def get_categories(self) -> List[MarketplaceCategory]: + """Return the predefined marketplace categories with live skill counts.""" + self._ensure_loaded() + counts: Dict[str, int] = {} + for entry in self._entries: + for cat in entry.categories: + counts[cat] = counts.get(cat, 0) + 1 + return [ + MarketplaceCategory( + name=c.name, + description=c.description, + icon=c.icon, + skill_count=counts.get(c.name, 0), + ) + for c in _DEFAULT_CATEGORIES + ] + + # ── Discovery API ───────────────────────────────────────────────────── + + def get_featured(self) -> List[MarketplaceEntry]: + """Return entries marked as featured or trending.""" + self._ensure_loaded() + return [e for e in self._entries if e.featured or e.trending] + + def get_by_category(self, category: str) -> List[MarketplaceEntry]: + """Filter entries belonging to *category*.""" + self._ensure_loaded() + return [e for e in self._entries if category in e.categories] + + def get_trending(self, limit: int = 10) -> List[MarketplaceEntry]: + """Top skills sorted by recent installs / downloads.""" + self._ensure_loaded() + ranked = sorted( + self._entries, + key=lambda e: (e.install_count, e.downloads), + reverse=True, + ) + return ranked[:limit] + + def search( + self, + query: str, + category: Optional[str] = None, + ) -> List[MarketplaceEntry]: + """Enhanced local search with optional category filter and relevance ranking.""" + self._ensure_loaded() + q = query.lower() + candidates = self._entries + if category: + candidates = [e for e in candidates if category in e.categories] + + def _score(entry: MarketplaceEntry) -> float: + score = 0.0 + if q in entry.name.lower(): + score += 10.0 + if q in entry.description.lower(): + score += 5.0 + if entry.featured: + score += 3.0 + if entry.trending: + score += 2.0 + score += min(entry.downloads / 1000.0, 5.0) + return score + + scored = [(e, _score(e)) for e in candidates] + scored = [(e, s) for e, s in scored if s > 0] + scored.sort(key=lambda t: t[1], reverse=True) + return [e for e, _ in scored] + + # ── Install ─────────────────────────────────────────────────────────── + + async def install(self, name_or_entry: str | MarketplaceEntry) -> str: + """Install a skill from the marketplace via hub.pull(). + + Records install statistics locally. + + Args: + name_or_entry: Either a skill name / repo_id string or a + MarketplaceEntry. + + Returns: + Human-readable status message. + """ + if isinstance(name_or_entry, MarketplaceEntry): + if name_or_entry.hub_type: + repo_id = f"{name_or_entry.hub_type}://{name_or_entry.repo_id}" + else: + repo_id = name_or_entry.repo_id + display = name_or_entry.name + else: + repo_id = name_or_entry + display = name_or_entry + + try: + bundle = await self._hub.pull(repo_id) + self._record_install(display) + return ( + f"Installed '{bundle.manifest.name}' " + f"v{bundle.manifest.version} from {repo_id}." + ) + except Exception as exc: + return f"Install failed for '{display}': {type(exc).__name__}: {exc}" + + # ── Refresh / Cache ─────────────────────────────────────────────────── + + async def refresh(self) -> int: + """Fetch latest catalog from hub backends and update local cache. + + Returns: + Number of entries in the refreshed catalog. + """ + new_entries: List[MarketplaceEntry] = [] + try: + results = await self._hub.federated_search("") + summaries: Sequence[SkillSummary] = ( + list(results.results) if hasattr(results, "results") else results + ) + for s in summaries: + cats = self._infer_categories(s) + entry = MarketplaceEntry.from_summary( + s, + categories=tuple(cats), + install_count=self._install_stats.get(s.name, 0), + trending=s.downloads > 50, + ) + new_entries.append(entry) + except Exception as exc: + logger.warning("Marketplace refresh from hub failed: %s", exc) + + if new_entries: + self._entries = new_entries + self._save_cache() + self._loaded = True + return len(self._entries) + + # ── Private Helpers ─────────────────────────────────────────────────── + + def _ensure_loaded(self) -> None: + """Load from local cache if not already loaded.""" + if self._loaded: + return + self._load_cache() + self._loaded = True + + def _load_cache(self) -> None: + """Read marketplace_cache.json from disk.""" + if self._cache_path is None or not self._cache_path.exists(): + return + try: + raw = json.loads(self._cache_path.read_text(encoding="utf-8")) + entries = raw.get("entries", []) + if not isinstance(entries, list): + logger.warning("Marketplace cache entries malformed; ignoring cache") + return + + for item in entries: + # Convert categories list to tuple for frozen dataclass + if isinstance(item, dict) and isinstance(item.get("categories"), list): + item["categories"] = tuple(item["categories"]) + try: + self._entries.append(MarketplaceEntry(**item)) + except (TypeError, AttributeError): + logger.debug("Skipping malformed cache entry: %r", item) + self._install_stats = raw.get("install_stats", {}) + except (json.JSONDecodeError, OSError) as exc: + logger.warning("Failed to load marketplace cache: %s", exc) + + def _save_cache(self) -> None: + """Persist current entries and install stats to disk.""" + if self._cache_path is None: + return + try: + self._cache_path.parent.mkdir(parents=True, exist_ok=True) + data = { + "version": 1, + "updated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + "entries": [_entry_to_dict(e) for e in self._entries], + "install_stats": self._install_stats, + } + self._cache_path.write_text( + json.dumps(data, indent=2, ensure_ascii=False), + encoding="utf-8", + ) + except OSError as exc: + logger.warning("Failed to save marketplace cache: %s", exc) + + def _record_install(self, name: str) -> None: + """Increment local install counter and persist.""" + self._install_stats[name] = self._install_stats.get(name, 0) + 1 + self._save_cache() + + @staticmethod + def _infer_categories(summary: SkillSummary) -> List[str]: + """Heuristic category inference from skill name/description.""" + text = f"{summary.name} {summary.description}".lower() + cats: List[str] = [] + _KEYWORD_MAP: Dict[str, List[str]] = { + "development": ["code", "develop", "build", "compile", "debug", "lint", "refactor"], + "research": ["research", "paper", "academic", "experiment", "science"], + "automation": ["automat", "workflow", "pipeline", "schedule", "batch"], + "productivity": ["productiv", "organiz", "note", "calendar", "todo"], + "security": ["secur", "vulnerab", "audit", "scan", "cve", "encrypt"], + "operations": ["devops", "deploy", "monitor", "infra", "docker", "k8s", "ci/cd"], + "analysis": ["analy", "data", "visual", "chart", "statistic", "metric"], + "integration": ["integrat", "connect", "bridge", "api", "webhook", "sync"], + } + for cat, keywords in _KEYWORD_MAP.items(): + if any(kw in text for kw in keywords): + cats.append(cat) + return cats or ["development"] + + +def _entry_to_dict(entry: MarketplaceEntry) -> Dict[str, Any]: + """Serialize a MarketplaceEntry to a JSON-safe dict.""" + d = asdict(entry) + # Ensure categories is a list for JSON serialization + d["categories"] = list(d.get("categories", ())) + return d diff --git a/src/leapflow/plugins/tool_plugins/skill_discovery.py b/src/leapflow/plugins/tool_plugins/skill_discovery.py index 0d490894..dafd8ca2 100644 --- a/src/leapflow/plugins/tool_plugins/skill_discovery.py +++ b/src/leapflow/plugins/tool_plugins/skill_discovery.py @@ -3,7 +3,7 @@ from __future__ import annotations -from leapflow.skills.discovery import skill_view, skills_list +from leapflow.skills.discovery import skill_file_read, skill_list_files, skill_view, skills_list from leapflow.plugins.protocol import ToolMetadata @@ -56,6 +56,38 @@ def tools(self) -> list[ToolMetadata]: x_leapflow={"category": "read", "plane": "task"}, provides_capabilities=("skill.view",), ), + ToolMetadata( + name="skill_list_files", + description="List all support files in a skill directory (configs, templates, examples). Use before skill_file_read to discover available files.", + parameters_schema={ + "type": "object", + "properties": { + "name": {"type": "string", "description": "Skill name to list files for"}, + }, + "required": ["name"], + }, + handler=skill_list_files, + x_leapflow={"category": "skills", "risk_level": "none", "plane": "task"}, + provides_capabilities=("skill.list_files",), + ), + ToolMetadata( + name="skill_file_read", + description="Read an individual support file from a skill directory. Use skill_list_files first to discover available files.", + parameters_schema={ + "type": "object", + "properties": { + "name": {"type": "string", "description": "Skill name"}, + "file_path": { + "type": "string", + "description": "Relative path within the skill directory (no '..' or absolute paths)", + }, + }, + "required": ["name", "file_path"], + }, + handler=skill_file_read, + x_leapflow={"category": "skills", "risk_level": "none", "plane": "task"}, + provides_capabilities=("skill.file_read",), + ), ] @property diff --git a/src/leapflow/skills/__init__.py b/src/leapflow/skills/__init__.py index e8552a72..eaba6906 100644 --- a/src/leapflow/skills/__init__.py +++ b/src/leapflow/skills/__init__.py @@ -2,6 +2,8 @@ """Skills package — runtime skill registry, activation, and execution.""" from leapflow.skills.curator import ( + ConsolidationAction, + ConsolidationSuggestion, CurationReport, CurationState, CurationTransition, @@ -10,6 +12,7 @@ ) from leapflow.skills.index import SkillEntry, SkillIndex from leapflow.skills.injector import SkillInjector +from leapflow.skills.mcp_bridge import McpSkillBridge, McpSkillEntry from leapflow.skills.registry import ( Skill, SkillMetadata, @@ -19,9 +22,13 @@ ) __all__ = [ + "ConsolidationAction", + "ConsolidationSuggestion", "CurationReport", "CurationState", "CurationTransition", + "McpSkillBridge", + "McpSkillEntry", "Skill", "SkillCurationEntry", "SkillCurator", diff --git a/src/leapflow/skills/action_policy.py b/src/leapflow/skills/action_policy.py index 80ac001c..29243bd0 100644 --- a/src/leapflow/skills/action_policy.py +++ b/src/leapflow/skills/action_policy.py @@ -24,7 +24,7 @@ from enum import Enum from typing import List, Optional, Protocol, runtime_checkable -from leapflow.skills.tool_executor import ToolCall +from leapflow.skills.tool_types import ToolCall class Verdict(Enum): diff --git a/src/leapflow/skills/builtin_skills/.skills_index.json b/src/leapflow/skills/builtin_skills/.skills_index.json new file mode 100644 index 00000000..62ff71a1 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/.skills_index.json @@ -0,0 +1 @@ +[{"name": "api_designer", "description": "REST and GraphQL API design with OpenAPI spec generation and best practices", "category": "development", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["API", "REST", "GraphQL", "OpenAPI", "design", "specification"], "requires_tools": ["file_read", "file_write"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/api_designer"}, {"name": "code_review", "description": "Structured code review for security, performance, and style", "category": "development", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["code", "review", "security", "performance", "quality"], "requires_tools": ["file_read", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/code_review"}, {"name": "data_analysis", "description": "CSV/JSON data analysis with statistical summaries, quality checks, and visualization guidance", "category": "analysis", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["data", "analysis", "statistics", "CSV", "JSON", "visualization"], "requires_tools": ["file_read", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/data_analysis"}, {"name": "devops_helper", "description": "CI/CD pipeline configuration, Docker orchestration, and deployment strategy", "category": "operations", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["devops", "CI/CD", "Docker", "deployment", "pipeline", "infrastructure"], "requires_tools": ["file_read", "file_write", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/devops_helper"}, {"name": "document_writer", "description": "Structured document generation: technical docs, API references, reports, and READMEs", "category": "productivity", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["documentation", "technical-writing", "README", "report", "API-docs"], "requires_tools": ["file_read", "file_write"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/document_writer"}, {"name": "git_workflow", "description": "Git operation orchestration: branching, commits, conflicts, and PR workflow", "category": "development", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["git", "version-control", "branching", "commits", "pull-request", "conflict-resolution"], "requires_tools": ["shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/git_workflow"}, {"name": "i18n_helper", "description": "Internationalization workflow: string extraction, translation, locale management, and validation", "category": "productivity", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["i18n", "internationalization", "localization", "translation", "l10n", "locale"], "requires_tools": ["file_read", "file_write", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/i18n_helper"}, {"name": "project_analysis", "description": "Project structure analysis, dependency audit, and tech stack identification", "category": "development", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["project", "analysis", "dependencies", "architecture", "tech-stack"], "requires_tools": ["file_read", "file_list", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/project_analysis"}, {"name": "security_audit", "description": "Security vulnerability scanning strategy, threat modeling, and fix recommendations", "category": "security", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["security", "vulnerability", "audit", "OWASP", "threat-model", "hardening"], "requires_tools": ["file_read", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/security_audit"}, {"name": "shell_automation", "description": "Natural language to shell commands with safety review and step-by-step execution", "category": "automation", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["shell", "terminal", "automation", "commands", "scripting"], "requires_tools": ["shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/shell_automation"}, {"name": "test_writer", "description": "Generate unit and integration tests from source code with edge-case coverage", "category": "development", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["testing", "unit-test", "integration-test", "test-generation", "coverage"], "requires_tools": ["file_read", "file_write", "shell_run"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/test_writer"}, {"name": "web_research", "description": "Multi-step web search with cross-verification and structured synthesis", "category": "research", "source": "builtin", "confidence": 1.0, "quality_score": 1.0, "tags": ["web", "search", "research", "synthesis", "fact-checking"], "requires_tools": ["web_search", "web_fetch"], "platforms": [], "skill_dir": "src/leapflow/skills/builtin_skills/web_research"}] \ No newline at end of file diff --git a/src/leapflow/skills/builtin_skills/api_designer/SKILL.md b/src/leapflow/skills/builtin_skills/api_designer/SKILL.md new file mode 100644 index 00000000..1f1270d9 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/api_designer/SKILL.md @@ -0,0 +1,189 @@ +--- +name: api_designer +description: "REST and GraphQL API design with OpenAPI spec generation and best practices" +version: 1.0.0 +metadata: + leapflow: + category: "development" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "development" + tags: ["API", "REST", "GraphQL", "OpenAPI", "design", "specification"] + requires_tools: ["file_read", "file_write"] +platforms: [] +triggers: + - "design API" + - "OpenAPI" + - "REST API" + - "GraphQL" + - "API设计" + - "接口设计" + - "API specification" + - "endpoint design" +--- + +# API Designer + +## Purpose + +Design clean, consistent, and evolvable APIs — REST or GraphQL — and produce +machine-readable specifications (OpenAPI 3.1, GraphQL SDL). This skill treats +API design as a contract negotiation between producer and consumer, prioritizing +developer experience, backward compatibility, and operational clarity. + +## Guiding Principles + +1. **Contract first** — Design the API before writing implementation code. + The specification is the source of truth; code conforms to it. +2. **Resource-oriented thinking** — Model APIs around domain nouns (resources), + not implementation verbs. `/orders/{id}` not `/getOrderById`. +3. **Consistency is kindness** — Naming, pagination, error format, and + authentication must be uniform across every endpoint. +4. **Evolvability over perfection** — Design for extension (additive changes) + without breaking existing clients. Avoid enums in responses; prefer + open-ended strings with documented values. +5. **Operational transparency** — Every API must be traceable (request IDs), + rate-limited, and versioned. + +## Workflow + +### Phase 1 — Understand Requirements + +Before designing endpoints: + +1. Identify the **domain model**: what are the core entities and their + relationships? +2. Determine the **consumers**: frontend SPA, mobile app, third-party + integrators, internal microservices. Different consumers imply different + granularity and authentication. +3. List the **use cases**: what actions do consumers need to perform? Map each + to a resource + operation. +4. Clarify **constraints**: auth method (OAuth2, API key, JWT), rate limits, + data sensitivity, compliance requirements (GDPR, HIPAA). +5. Read existing API code or specs to maintain consistency. + +### Phase 2 — Resource Modeling + +Map domain entities to REST resources: + +- **Naming**: plural nouns, lowercase, hyphen-separated + (`/order-items`, not `/orderItems` or `/OrderItem`). +- **Hierarchy**: nest only for strong ownership (`/users/{id}/addresses`); + use query parameters for loose associations. +- **Operations**: + + | Action | Method | Path | Status | + |---|---|---|---| + | List | GET | `/resources` | 200 | + | Create | POST | `/resources` | 201 | + | Read | GET | `/resources/{id}` | 200 | + | Full update | PUT | `/resources/{id}` | 200 | + | Partial update | PATCH | `/resources/{id}` | 200 | + | Delete | DELETE | `/resources/{id}` | 204 | + +- **Sub-resources vs query params**: use sub-resources for composition + (`/orders/{id}/line-items`); use query params for filtering, sorting, + and pagination on collection endpoints. + +For **GraphQL**: +- Design a schema with clear type boundaries. +- Use `input` types for mutations; never reuse output types as inputs. +- Implement connections (Relay-style cursor pagination) for lists. +- Keep resolvers thin; business logic lives in service layer. + +### Phase 3 — Standard Patterns + +Apply consistent patterns across the API: + +**Pagination** (choose one and use everywhere): +```json +{ + "data": [...], + "pagination": { + "cursor": "eyJpZCI6MTAwfQ==", + "has_more": true, + "total_count": 1523 + } +} +``` +Prefer cursor-based for large/changing datasets; offset-based for small, +stable datasets. + +**Filtering and sorting**: +``` +GET /orders?status=shipped&sort=-created_at&fields=id,status,total +``` +- Filter: `field=value` for equality; `field[gte]=10` for ranges. +- Sort: comma-separated fields; `-` prefix for descending. +- Sparse fields: `fields=` parameter to reduce payload. + +**Error format** (RFC 7807 Problem Details): +```json +{ + "type": "https://api.example.com/errors/insufficient-funds", + "title": "Insufficient Funds", + "status": 422, + "detail": "Account balance ($10.00) is below the required $25.00.", + "instance": "/orders/42" +} +``` + +**Versioning**: prefer URL-path versioning (`/v1/resources`) for simplicity; +use header versioning (`Accept: application/vnd.api+json;version=2`) when +URL changes are unacceptable. + +**Rate limiting headers**: +``` +X-RateLimit-Limit: 1000 +X-RateLimit-Remaining: 998 +X-RateLimit-Reset: 1625097600 +Retry-After: 30 +``` + +### Phase 4 — Generate OpenAPI Specification + +Produce an OpenAPI 3.1 YAML document: + +1. Define `info` (title, version, description, contact). +2. Define `servers` (dev, staging, production URLs). +3. Define `paths` with full request/response schemas. +4. Define `components/schemas` for all domain objects (use `$ref`). +5. Define `components/securitySchemes` and apply globally or per-operation. +6. Add `examples` for every endpoint — at least one success and one error. +7. Use `tags` to group related endpoints. + +Validate the spec: +``` +npx @redocly/cli lint openapi.yaml +``` + +### Phase 5 — Review Checklist + +Before finalizing: + +- [ ] Every endpoint has a description and at least one example. +- [ ] All 4xx/5xx responses are documented with error schemas. +- [ ] Pagination, filtering, and sorting are consistent across collections. +- [ ] Authentication is defined and applied to all non-public endpoints. +- [ ] No breaking changes if this is an API revision (additive only). +- [ ] Response bodies do not expose internal IDs, stack traces, or secrets. +- [ ] Enum values in responses use strings, not integers. + +## Error Handling + +| Situation | Action | +|---|---| +| Conflicting naming conventions in existing API | Document the conflict; follow the dominant pattern; suggest a migration plan for outliers. | +| Requirement implies RPC-style action (e.g., "send email") | Model as a resource creation: `POST /emails` with a body, not `POST /sendEmail`. | +| GraphQL schema grows too large | Split into modules/namespaces; use schema stitching or federation for microservices. | +| Consumer needs real-time updates | Recommend WebSocket or SSE alongside REST; define event schema. | + +## Limitations + +- This skill produces specifications and design documents, not running server + code. Implementation scaffolding requires the target framework's CLI. +- API security testing (penetration testing, fuzzing) is out of scope; the + skill focuses on design-time security patterns. +- Mock server generation depends on external tools (Prism, WireMock). diff --git a/src/leapflow/skills/builtin_skills/code_review/SKILL.md b/src/leapflow/skills/builtin_skills/code_review/SKILL.md new file mode 100644 index 00000000..787f238c --- /dev/null +++ b/src/leapflow/skills/builtin_skills/code_review/SKILL.md @@ -0,0 +1,168 @@ +--- +name: code_review +description: "Structured code review for security, performance, and style" +version: 1.0.0 +metadata: + leapflow: + category: "development" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "development" + tags: ["code", "review", "security", "performance", "quality"] + requires_tools: ["file_read", "shell_run"] +platforms: [] +triggers: + - "review this code" + - "code review" + - "check this code for issues" + - "audit this code" + - "review my changes" + - "find bugs in this code" + - "security review" + - "review pull request" +--- + +# Code Review + +## Purpose + +Perform a thorough, structured code review that evaluates changes across four +dimensions: correctness, security, performance, and maintainability. Produce +actionable feedback organized by severity so the author can prioritize fixes. + +## Guiding Principles + +1. **Understand intent first** — Before critiquing implementation, understand + what the code is trying to achieve. Read surrounding context, commit + messages, and related files. +2. **Severity matters** — Distinguish between blocking issues (bugs, security + holes) and suggestions (style, naming). Never bury a critical finding in + a list of nitpicks. +3. **Be specific and actionable** — "This is bad" is not feedback. Always + explain *why* something is a problem and *how* to fix it. +4. **Assume competence** — The author made deliberate choices. When something + looks wrong, consider whether there is a reason before flagging it. +5. **Scope discipline** — Review the change, not the entire codebase. Flag + pre-existing issues only when the change makes them worse or when they + create a direct interaction risk. + +## Workflow + +### Phase 1 — Scope the Review + +Determine what changed and establish context: + +1. Identify the **files changed** — use `file_read` to examine each file or + use `shell_run` with `git diff` to get the change set. +2. Read **surrounding code** for each changed function/class to understand + the integration surface. +3. Check for a **test file** corresponding to each changed source file. +4. Note the **language, framework, and project conventions** in use. + +Produce a brief scope summary: +- N files changed, M lines added/removed +- Primary area: +- Languages: + +### Phase 2 — Correctness Analysis + +Walk through the logic of each change: + +- Does the code do what it claims to do? +- Are **edge cases** handled (null/empty inputs, boundary values, error paths)? +- Is **error handling** present and appropriate? Are exceptions caught at the + right granularity? +- Do **types and contracts** match (function signatures, return types, API + schemas)? +- If tests exist, do they cover the new/changed paths? Are assertions + meaningful? + +### Phase 3 — Security Analysis + +Examine each change for common vulnerability patterns: + +- **Injection**: SQL, command, template, XSS — is user input sanitized before + use in queries, shell commands, or rendered output? +- **Authentication / Authorization**: Does the change bypass or weaken + access controls? Are secrets hardcoded? +- **Data exposure**: Could the change leak sensitive data in logs, error + messages, or API responses? +- **Deserialization**: Is untrusted data deserialized without validation? +- **Dependencies**: Are new dependencies introduced? Are they from trusted + sources, actively maintained, and free of known CVEs? +- **Concurrency**: Race conditions, TOCTOU, shared mutable state without + synchronization. + +For each finding, assess exploitability (not just theoretical possibility). + +### Phase 4 — Performance Analysis + +Look for patterns that could degrade runtime or resource usage: + +- **Algorithmic complexity**: O(n²) loops, unnecessary repeated computation, + missing caches for expensive operations. +- **I/O patterns**: Unbounded queries, N+1 database calls, missing pagination, + synchronous blocking in async code. +- **Memory**: Large allocations in hot paths, unbounded collections, missing + resource cleanup. +- **Concurrency**: Lock contention, excessive context switching, thread-unsafe + shared state. + +Flag only issues that are *likely* to matter at the project's scale. + +### Phase 5 — Maintainability & Style + +Evaluate code quality and long-term health: + +- **Naming**: Are variables, functions, and classes named clearly and + consistently? +- **Structure**: Is the code well-factored? Are responsibilities separated? +- **Duplication**: Is there copy-paste that should be extracted? +- **Documentation**: Are public APIs documented? Are complex algorithms + explained? +- **Consistency**: Does the change follow the project's existing conventions? + +Style issues are lowest priority — flag them, but clearly separate from +blocking concerns. + +### Phase 6 — Report + +Produce a structured review: + +``` +## Review Summary +<1–3 sentence overall assessment: approve / request changes / needs discussion> + +## Critical Issues (must fix) +1. **[SECURITY]** : — + **Fix**: + +## Warnings (should fix) +1. **[PERF]** : — + +## Suggestions (nice to have) +1. **[STYLE]** : — + +## Positive Notes +- +``` + +Rules: +- Categorize every finding: `[BUG]`, `[SECURITY]`, `[PERF]`, `[STYLE]`, + `[TEST]`, `[DOC]`. +- Include file path and line number (or function name) for each finding. +- "Critical" = the change should not merge without addressing this. +- "Warning" = strongly recommended but not blocking. +- "Suggestion" = optional improvement. +- Always end with something positive — reinforce good patterns. + +## Error Handling + +| Situation | Action | +|---|---| +| Cannot read a changed file | Report which files were inaccessible; review what you can. | +| No test files found | Flag as a warning: "No corresponding tests found for ." | +| Unfamiliar language or framework | State your confidence level; focus on universal principles (logic, security, naming). | +| Change is too large (>1000 lines) | Suggest splitting the review; focus on the highest-risk files first. | diff --git a/src/leapflow/skills/builtin_skills/data_analysis/SKILL.md b/src/leapflow/skills/builtin_skills/data_analysis/SKILL.md new file mode 100644 index 00000000..79f11b66 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/data_analysis/SKILL.md @@ -0,0 +1,193 @@ +--- +name: data_analysis +description: "CSV/JSON data analysis with statistical summaries, quality checks, and visualization guidance" +version: 1.0.0 +metadata: + leapflow: + category: "analysis" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "analysis" + tags: ["data", "analysis", "statistics", "CSV", "JSON", "visualization"] + requires_tools: ["file_read", "shell_run"] +platforms: [] +triggers: + - "analyze data" + - "data analysis" + - "CSV analysis" + - "statistics" + - "数据分析" + - "统计分析" + - "summarize this data" + - "data quality check" +--- + +# Data Analysis + +## Purpose + +Analyze structured data (CSV, JSON, TSV, Parquet) through a systematic pipeline: +inspect the data, assess quality, compute descriptive statistics, detect patterns, +and produce actionable insights with visualization recommendations. This skill +turns raw data into understanding. + +## Guiding Principles + +1. **Look before you compute** — Always inspect raw data (head, tail, shape, + dtypes) before running any analysis. Assumptions about structure are the + leading cause of wrong results. +2. **Quality gates** — Missing values, duplicates, type mismatches, and encoding + errors must be identified and reported before analysis proceeds. +3. **Context over numbers** — A mean is meaningless without understanding what + the column represents. Always relate statistics back to the domain. +4. **Appropriate methods** — Choose statistical measures that match the data + distribution. Median and IQR for skewed data; mean and std for normal. +5. **Reproducibility** — Every analysis step must be scriptable and repeatable. + Provide the exact commands or code used. + +## Workflow + +### Phase 1 — Data Ingestion and Inspection + +1. Read the file with `file_read` to examine the first 20–50 rows and understand + structure. +2. Determine format: CSV (detect delimiter, quoting, encoding), JSON (flat vs + nested), TSV, or other. +3. Record: + - **Shape**: number of rows × columns. + - **Columns**: name, inferred type (numeric, categorical, datetime, text). + - **Sample values**: 3–5 example values per column. +4. If the file is large (>10k rows), use `shell_run` with Python or + command-line tools for efficient processing: + ``` + python3 -c "import pandas as pd; df=pd.read_csv('data.csv'); print(df.shape); print(df.dtypes); print(df.head())" + ``` + +### Phase 2 — Data Quality Assessment + +Evaluate data health before analysis: + +- **Missing values**: count per column, percentage, pattern (random vs + systematic — e.g., all missing for certain dates or categories). +- **Duplicates**: exact row duplicates and near-duplicates on key columns. +- **Type issues**: numeric columns stored as strings, inconsistent date formats, + mixed types within a column. +- **Outliers**: values beyond 3σ or 1.5×IQR from the median — flag but do not + remove without domain justification. +- **Encoding**: check for mojibake, BOM markers, or mixed encodings. + +Produce a quality summary: +``` +Data Quality Report: + Rows: 15,234 | Columns: 12 + Missing: 3 columns with >5% missing (col_a: 12%, col_b: 7%, col_c: 6%) + Duplicates: 43 exact duplicates found + Type issues: col_price has 18 non-numeric entries + Outliers: col_age has 5 values > 120 +``` + +Recommend cleaning actions for each issue found. + +### Phase 3 — Descriptive Statistics + +Compute statistics appropriate to each column type: + +**Numeric columns**: +- Central tendency: mean, median, mode. +- Dispersion: std, variance, IQR, range. +- Shape: skewness, kurtosis. +- Quantiles: 5th, 25th, 50th, 75th, 95th percentiles. + +**Categorical columns**: +- Unique count and cardinality ratio (unique/total). +- Top-N value frequency (with percentages). +- Rare categories (appearing < 1% of rows). + +**Datetime columns**: +- Range (min to max), span. +- Frequency/granularity (daily, monthly, irregular). +- Gaps: missing dates in an otherwise regular series. + +**Cross-column**: +- Correlation matrix for numeric pairs (flag |r| > 0.7). +- Contingency tables for categorical pairs when relevant. + +### Phase 4 — Pattern Detection and Insights + +Go beyond summary statistics: + +1. **Trends**: for time-series data, identify upward/downward trends, + seasonality, and change points. +2. **Segmentation**: group by categorical columns and compare numeric + distributions across groups. +3. **Anomalies**: data points that are statistically unusual and may indicate + errors, fraud, or interesting phenomena. +4. **Relationships**: notable correlations, dependencies, or interactions + between columns. + +Each insight must include: +- **What** was found (specific numbers). +- **Why** it might matter (domain interpretation). +- **Confidence level** (strong evidence vs. suggestive pattern). + +### Phase 5 — Visualization Recommendations + +For each key finding, recommend the most effective chart: + +| Data Pattern | Recommended Chart | +|---|---| +| Distribution of one variable | Histogram or KDE plot | +| Comparison across categories | Bar chart (horizontal for many categories) | +| Trend over time | Line chart with confidence band | +| Relationship between two numerics | Scatter plot with regression line | +| Part-of-whole composition | Stacked bar or treemap (not pie) | +| Multivariate relationships | Heatmap (correlation) or parallel coordinates | +| Outlier detection | Box plot or violin plot | + +Provide ready-to-run code (matplotlib/seaborn or the project's preferred library) +for the top 3 recommended visualizations. + +### Phase 6 — Report + +Produce a structured analysis report: + +``` +## Dataset Overview + + +## Data Quality + + +## Key Statistics + + +## Insights +1. +2. ... + +## Recommended Visualizations + + +## Next Steps + +``` + +## Error Handling + +| Situation | Action | +|---|---| +| File too large for memory | Use chunked reading (`chunksize` in pandas) or sample first 10k rows with a disclaimer. | +| Encoding errors | Try UTF-8, then Latin-1, then detect with `chardet`; report encoding used. | +| All numeric columns are actually IDs | Flag that statistical summaries are meaningless for ID columns; exclude from analysis. | +| No clear structure (freeform text file) | Report that the file is not structured tabular data; suggest NLP-based analysis instead. | + +## Limitations + +- This skill does not perform machine learning (regression, classification). + It provides the exploratory analysis that informs modeling decisions. +- Visualization code is generated but not rendered; the user must execute it + in their environment. +- Very large datasets (>100M rows) may require distributed tools beyond + the scope of this skill. diff --git a/src/leapflow/skills/builtin_skills/devops_helper/SKILL.md b/src/leapflow/skills/builtin_skills/devops_helper/SKILL.md new file mode 100644 index 00000000..4b7f1aeb --- /dev/null +++ b/src/leapflow/skills/builtin_skills/devops_helper/SKILL.md @@ -0,0 +1,177 @@ +--- +name: devops_helper +description: "CI/CD pipeline configuration, Docker orchestration, and deployment strategy" +version: 1.0.0 +metadata: + leapflow: + category: "operations" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "operations" + tags: ["devops", "CI/CD", "Docker", "deployment", "pipeline", "infrastructure"] + requires_tools: ["file_read", "file_write", "shell_run"] +platforms: [] +triggers: + - "CI/CD" + - "Docker" + - "deploy" + - "DevOps" + - "pipeline" + - "部署" + - "持续集成" + - "container orchestration" +--- + +# DevOps Helper + +## Purpose + +Generate and troubleshoot CI/CD pipelines, Docker configurations, and deployment +strategies. This skill bridges the gap between application code and the +infrastructure that builds, tests, and ships it — producing configurations that +are secure, reproducible, and maintainable. + +## Guiding Principles + +1. **Infrastructure as code** — Every configuration must be version-controlled, + reviewable, and reproducible. No manual console clicks. +2. **Least privilege** — Containers run as non-root, CI tokens have minimal + scopes, secrets never appear in logs or images. +3. **Fail fast, fail loud** — Pipelines should catch errors early and report + them clearly. A green build must mean the artifact is shippable. +4. **Immutable artifacts** — Build once, deploy many times. The same image + that passes staging goes to production. +5. **Incremental complexity** — Start with the simplest pipeline that works; + add caching, parallelism, and matrix builds only when justified. + +## Workflow + +### Phase 1 — Assess the Project + +1. Read project structure to understand: + - Language and build system (package.json, pyproject.toml, go.mod, Makefile). + - Existing CI config (.github/workflows/, .gitlab-ci.yml, Jenkinsfile, + .circleci/). + - Existing Docker files (Dockerfile, docker-compose.yml, .dockerignore). + - Deployment targets (cloud provider, Kubernetes, bare metal, serverless). +2. Identify the **deployment pipeline stages** already in place vs missing: + - Build → Test → Lint → Security scan → Package → Deploy → Smoke test. +3. Note environment-specific requirements: secrets, environment variables, + database migrations, feature flags. + +### Phase 2 — CI/CD Pipeline Design + +Generate or improve CI/CD configuration: + +**GitHub Actions** (default when `.github/` exists): +- Use reusable workflows for shared steps. +- Pin action versions to SHA, not tags. +- Separate jobs for lint, test, build, deploy — with explicit dependencies. +- Cache dependencies (`actions/cache`) keyed on lockfile hash. +- Use matrix strategy for multi-version testing. +- Set `concurrency` to cancel superseded runs on the same branch. + +**Pipeline structure template**: +```yaml +name: CI +on: + push: + branches: [main] + pull_request: + branches: [main] + +jobs: + lint: + runs-on: ubuntu-latest + steps: [checkout, setup, lint] + + test: + runs-on: ubuntu-latest + strategy: + matrix: + version: ["3.11", "3.12"] + steps: [checkout, setup, install-deps, run-tests, upload-coverage] + + build: + needs: [lint, test] + steps: [checkout, build-artifact, upload-artifact] + + deploy: + needs: [build] + if: github.ref == 'refs/heads/main' + environment: production + steps: [download-artifact, deploy] +``` + +Adapt to the project's actual CI platform and requirements. + +### Phase 3 — Docker Configuration + +Generate production-grade Dockerfiles: + +1. **Multi-stage builds**: separate build and runtime stages to minimize + image size. +2. **Base image selection**: use official slim/alpine images; pin to specific + digest or version tag. +3. **Layer ordering**: copy dependency manifests first, install dependencies, + then copy source — maximizes cache reuse. +4. **Security**: + - Run as non-root user (create and switch with `USER`). + - No secrets in build args or layers. + - Scan with `docker scout` or `trivy`. +5. **Health checks**: include `HEALTHCHECK` for orchestrated deployments. +6. **`.dockerignore`**: exclude `.git/`, `node_modules/`, `__pycache__/`, + test fixtures, and documentation. + +**docker-compose.yml** for local development: +- Define services with proper dependency ordering (`depends_on` with + `condition: service_healthy`). +- Use named volumes for persistent data. +- Map ports explicitly; avoid `network_mode: host`. +- Provide `.env.example` for required environment variables. + +### Phase 4 — Deployment Strategy + +Recommend and implement a deployment strategy: + +| Strategy | When to Use | +|---|---| +| **Rolling update** | Default for stateless services; zero downtime, gradual rollout. | +| **Blue/green** | When instant rollback is critical; requires 2× resources briefly. | +| **Canary** | For high-traffic services; route small percentage to new version first. | +| **Feature flags** | When deployment and release should be decoupled. | +| **Recreate** | For stateful services that cannot run two versions simultaneously. | + +For each deployment: +- Define rollback triggers (error rate threshold, latency spike). +- Include smoke tests that run post-deploy. +- Ensure database migrations are backward-compatible for rolling deployments. + +### Phase 5 — Validation + +1. Lint all generated configs: + - YAML: `yamllint` or platform-specific validators. + - Dockerfile: `hadolint`. + - docker-compose: `docker compose config`. +2. Dry-run where possible (`act` for GitHub Actions, `docker build --check`). +3. Verify secrets are not hardcoded anywhere in the generated files. +4. Test the pipeline locally before pushing. + +## Error Handling + +| Situation | Action | +|---|---| +| Unknown CI platform | Generate GitHub Actions config and note how to adapt for other platforms. | +| Secrets required but not configured | Generate placeholder `${{ secrets.NAME }}` references and list all required secrets with setup instructions. | +| Docker build fails | Read the build output; common causes: missing system deps, wrong base image arch, cache invalidation. | +| Pipeline is slow (>15 min) | Audit for missing caches, unnecessary steps, serial jobs that could parallelize. | + +## Limitations + +- This skill generates configuration files; it does not have direct access to + CI/CD platforms or cloud consoles. +- Secret management setup (vault, cloud KMS) is guided but not automated. +- Kubernetes manifests are generated as static YAML; Helm chart generation + is out of scope. diff --git a/src/leapflow/skills/builtin_skills/document_writer/SKILL.md b/src/leapflow/skills/builtin_skills/document_writer/SKILL.md new file mode 100644 index 00000000..86254dfb --- /dev/null +++ b/src/leapflow/skills/builtin_skills/document_writer/SKILL.md @@ -0,0 +1,170 @@ +--- +name: document_writer +description: "Structured document generation: technical docs, API references, reports, and READMEs" +version: 1.0.0 +metadata: + leapflow: + category: "productivity" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "productivity" + tags: ["documentation", "technical-writing", "README", "report", "API-docs"] + requires_tools: ["file_read", "file_write"] +platforms: [] +triggers: + - "write document" + - "generate docs" + - "create README" + - "write report" + - "写文档" + - "生成文档" + - "technical documentation" + - "API documentation" +--- + +# Document Writer + +## Purpose + +Generate well-structured, audience-appropriate documents — technical documentation, +API references, project READMEs, architecture decision records, runbooks, and +reports. This skill treats documentation as a product: it must be correct, usable, +and maintained like code. + +## Guiding Principles + +1. **Audience first** — Identify who will read the document and what they need to + accomplish. An onboarding guide for new developers is not an API reference. +2. **Inverted pyramid** — Put the most important information first. Readers skim; + the answer should be findable in 30 seconds. +3. **Show, don't tell** — Concrete examples beat abstract descriptions. Every + non-trivial concept gets a code sample, diagram, or worked example. +4. **Single source of truth** — Documentation should live next to the code it + describes. Cross-references over duplication. +5. **Evergreen writing** — Avoid dates, version-specific language, and + "currently" phrasing that rots. Write for the next reader, not today's. + +## Workflow + +### Phase 1 — Analyze the Request + +Before generating any content, establish: + +1. **Document type**: README, API reference, architecture doc, runbook, + changelog, tutorial, or report. +2. **Audience**: end users, developers integrating an API, internal team members, + or stakeholders. Determine technical depth. +3. **Scope**: what must be covered, what is explicitly out of scope. +4. **Existing documentation**: read relevant existing docs to avoid contradiction + and find the right insertion point. +5. **Conventions**: check the project for documentation standards (file naming, + heading style, admonition syntax, link format). + +### Phase 2 — Structure the Document + +Choose a template based on the document type: + +**README**: +``` +# Project Name +> One-line description + +## Quick Start +## Features +## Installation +## Usage +## Configuration +## Contributing +## License +``` + +**API Reference**: +``` +# API Reference — +## Authentication +## Endpoints +### + - Description + - Parameters (table) + - Request example + - Response example + - Error codes +## Rate Limits +## Changelog +``` + +**Architecture Decision Record (ADR)**: +``` +# ADR-NNN: +## Status: <Proposed|Accepted|Deprecated|Superseded> +## Context +## Decision +## Consequences +## Alternatives Considered +``` + +**Runbook**: +``` +# Runbook: <Procedure> +## Prerequisites +## Steps (numbered, with verification after each) +## Rollback +## Contacts +``` + +Adapt the template to fit the project; never force content into a section that +adds no value. + +### Phase 3 — Generate Content + +For each section: + +1. Read the relevant source code, config files, or data with `file_read`. +2. Extract facts: function signatures, config keys, environment variables, + error codes, dependencies. +3. Write prose that is: + - **Concise**: one idea per paragraph, short sentences. + - **Precise**: use the exact names from the codebase (no paraphrasing + class names or API paths). + - **Active voice**: "The server starts on port 8080" not "Port 8080 is + used by the server for starting." +4. Add **code examples** for every usage pattern. Examples must be runnable + and syntactically correct. +5. Use **tables** for structured data (parameters, config keys, error codes). +6. Use **admonitions** (note, warning, tip) sparingly for critical information + that readers must not miss. + +### Phase 4 — Quality Check + +Before delivering: + +- **Accuracy**: do code examples actually work? Do file paths exist? +- **Completeness**: does every public API, config option, or workflow step + appear? +- **Consistency**: are heading levels, list styles, and code fence languages + uniform throughout? +- **Links**: are all cross-references valid? No broken relative paths. +- **Spelling and grammar**: proofread, especially proper nouns and technical + terms. + +Write the final document to the appropriate file with `file_write`. + +## Error Handling + +| Situation | Action | +|---|---| +| Source code is too complex to fully document | Focus on the public API surface; note internal details as out of scope. | +| Existing docs contradict the code | Trust the code. Update the docs and flag the discrepancy to the user. | +| No clear project conventions | Default to CommonMark, ATX headings, fenced code blocks, and sentence-case headings. | +| User requests a format you cannot render (PDF, Confluence) | Generate Markdown and advise on conversion tools (pandoc, markdown-to-confluence). | + +## Limitations + +- This skill generates Markdown (or plain text) documents. Rich formats + (PDF, DOCX, HTML) require external conversion. +- Diagram generation is descriptive (Mermaid code blocks); rendering depends + on the viewer. +- Accuracy depends on reading the codebase; if source files are inaccessible, + the document will have gaps. diff --git a/src/leapflow/skills/builtin_skills/git_workflow/SKILL.md b/src/leapflow/skills/builtin_skills/git_workflow/SKILL.md new file mode 100644 index 00000000..7bb25604 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/git_workflow/SKILL.md @@ -0,0 +1,171 @@ +--- +name: git_workflow +description: "Git operation orchestration: branching, commits, conflicts, and PR workflow" +version: 1.0.0 +metadata: + leapflow: + category: "development" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "development" + tags: ["git", "version-control", "branching", "commits", "pull-request", "conflict-resolution"] + requires_tools: ["shell_run"] +platforms: [] +triggers: + - "git workflow" + - "branch management" + - "commit convention" + - "resolve conflict" + - "git操作" + - "分支管理" + - "create pull request" + - "git best practices" +--- + +# Git Workflow + +## Purpose + +Orchestrate Git operations with disciplined branching strategy, consistent commit +conventions, systematic conflict resolution, and streamlined PR workflow. This +skill does not simply run git commands — it enforces a methodology that keeps the +repository history clean, bisectable, and reviewable. + +## Guiding Principles + +1. **History is documentation** — Every commit message is a permanent record read + by future developers. Treat it with the same care as code comments. +2. **Atomic commits** — Each commit captures exactly one logical change. A commit + that mixes a refactor with a feature is two commits. +3. **Branch hygiene** — Short-lived branches merged frequently beat long-lived + branches merged painfully. Delete merged branches immediately. +4. **Safety first** — Never rewrite published history. Use `--force-with-lease` + only on personal branches after explicit confirmation. +5. **Verify before sharing** — Every branch must build and pass tests locally + before pushing. + +## Workflow + +### Phase 1 — Assess the Situation + +Before running any git command, understand the current state: + +1. Run `git status` and `git log --oneline -10` to see working tree state and + recent history. +2. Identify the **branching model** in use: + - **Trunk-based**: short-lived feature branches off `main`, merged via PR. + - **GitFlow**: `develop` as integration branch, `release/*` and `hotfix/*` + branches for releases and urgent fixes. + - **Unknown**: inspect branch names and merge patterns to infer the model. +3. Check for uncommitted changes, stashed work, or in-progress rebases. +4. Confirm the **remote** topology (`git remote -v`). + +### Phase 2 — Branch Management + +Create or navigate branches following the project's model: + +- **Naming convention**: `<type>/<ticket>-<short-description>` + (e.g. `feat/PROJ-42-add-auth`, `fix/PROJ-99-null-pointer`). +- **Base branch**: always branch from the latest upstream target + (`git fetch origin && git checkout -b <branch> origin/main`). +- **Rebase vs merge**: prefer `git rebase` to keep a linear history on feature + branches; use `git merge --no-ff` when recording an explicit merge point. +- **Cleanup**: after merge, delete the local and remote branch + (`git branch -d <branch> && git push origin --delete <branch>`). + +### Phase 3 — Commit Conventions + +Apply Conventional Commits format: + +``` +<type>(<scope>): <subject> + +<body> + +<footer> +``` + +**Types**: `feat`, `fix`, `refactor`, `perf`, `test`, `docs`, `ci`, `chore`, +`style`, `build`. + +Rules: +- **Subject**: imperative mood, ≤72 characters, no trailing period. +- **Body**: wrap at 72 characters. Explain *why*, not *what* (the diff shows + what). Reference issue/ticket IDs. +- **Footer**: `BREAKING CHANGE:` for incompatible changes; `Refs:` or + `Closes:` for issue links. +- **Scope**: optional; matches the module or area affected. + +When staging changes: +1. Use `git add -p` for interactive staging to keep commits atomic. +2. Review the staged diff (`git diff --cached`) before committing. +3. Run `git commit` (not `git commit -m`) for multi-line messages when a body + is warranted. + +### Phase 4 — Conflict Resolution + +When merge or rebase conflicts arise: + +1. **Identify scope**: run `git diff --name-only --diff-filter=U` to list + conflicting files. +2. **Understand both sides**: for each conflict marker, read the surrounding + context to understand the intent of both changes. +3. **Resolution strategy**: + - **Ours-then-theirs**: when both changes are needed but ours should come + first (common in additive changes). + - **Theirs-wins**: when upstream refactored and our branch should adopt. + - **Manual merge**: when changes overlap semantically and require a new + combined implementation. +4. After resolving each file, run `git add <file>`. +5. Verify the resolution: run tests or at minimum a build check. +6. Complete with `git rebase --continue` or `git merge --continue`. + +Never blindly accept `--ours` or `--theirs` on the entire repository. + +### Phase 5 — PR Workflow + +Prepare and manage pull requests: + +1. **Pre-push checklist**: + - All tests pass locally. + - Linter/formatter has been run. + - Commit history is clean (squash fixups with `git rebase -i`). + - Branch is rebased on latest target. +2. **PR description template**: + ``` + ## What + <concise summary of the change> + + ## Why + <motivation, link to issue/ticket> + + ## How + <implementation approach, key decisions> + + ## Testing + <what was tested and how> + ``` +3. **Review cycle**: address feedback with fixup commits; squash before final + merge to keep the target branch clean. +4. **Merge method**: prefer "squash and merge" for single-purpose PRs; use + "merge commit" when preserving intermediate history matters. + +## Error Handling + +| Situation | Action | +|---|---| +| Detached HEAD state | Identify the intended branch; `git checkout <branch>` or create a new branch from current HEAD. | +| Accidental commit on wrong branch | `git cherry-pick` the commit to the correct branch, then `git reset` on the wrong one. | +| Force-push request | Refuse on shared branches. On personal branches, use `--force-with-lease` and confirm with the user first. | +| Large binary accidentally committed | Use `git filter-branch` or `git-filter-repo` to remove; advise `.gitignore` and LFS for future binaries. | +| Merge conflict during rebase | Resolve file-by-file as described in Phase 4; abort with `git rebase --abort` if the user requests. | + +## Limitations + +- This skill executes git commands via `shell_run`. Destructive operations + (force push, history rewrite, branch deletion) always require explicit user + confirmation before execution. +- Repository hosting platform APIs (GitHub, GitLab) are not directly accessible; + PR creation guidance is command-line oriented. diff --git a/src/leapflow/skills/builtin_skills/i18n_helper/SKILL.md b/src/leapflow/skills/builtin_skills/i18n_helper/SKILL.md new file mode 100644 index 00000000..f770311c --- /dev/null +++ b/src/leapflow/skills/builtin_skills/i18n_helper/SKILL.md @@ -0,0 +1,205 @@ +--- +name: i18n_helper +description: "Internationalization workflow: string extraction, translation, locale management, and validation" +version: 1.0.0 +metadata: + leapflow: + category: "productivity" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "productivity" + tags: ["i18n", "internationalization", "localization", "translation", "l10n", "locale"] + requires_tools: ["file_read", "file_write", "shell_run"] +platforms: [] +triggers: + - "internationalization" + - "i18n" + - "translate" + - "localization" + - "国际化" + - "翻译" + - "add language support" + - "extract strings" +--- + +# i18n Helper + +## Purpose + +Manage the full internationalization lifecycle: extract user-facing strings from +source code, organize them into locale files, produce translations, and validate +completeness across all supported languages. This skill treats i18n as a +structured engineering process — not an afterthought bolted onto finished code. + +## Guiding Principles + +1. **Extract, never hard-code** — Every user-visible string passes through the + i18n system. Literal strings in templates, error messages, or UI code are + defects. +2. **Key naming is API design** — Translation keys are the contract between code + and translators. Use semantic, hierarchical keys (`auth.login.button_label`) + not positional or arbitrary ones (`str_042`). +3. **Context for translators** — A key alone is not enough. Provide descriptions, + character limits, and screenshots where the string appears. +4. **Pluralization is not optional** — Different languages have different plural + rules (1 form for Chinese, 2 for English, 6 for Arabic). Use ICU + MessageFormat or the framework's plural system. +5. **Validate continuously** — Missing keys, unused keys, and format-string + mismatches must be caught in CI, not in production. + +## Workflow + +### Phase 1 — Audit the Current State + +1. Identify the **i18n framework** in use: + - JavaScript/TypeScript: `i18next`, `react-intl`, `vue-i18n`, `next-intl`. + - Python: `gettext`, `babel`, `django.utils.translation`. + - Mobile: `NSLocalizedString` (iOS), `strings.xml` (Android). + - None: the project has no i18n yet — skip to Phase 2b. +2. Locate **locale files**: `locales/`, `src/i18n/`, `messages/`, `*.po`, + `*.json`, `*.yaml`, `*.xliff`, `*.arb`. +3. Inventory **supported languages** and their completeness: + ``` + en: 342 keys (base) + zh-CN: 338 keys (4 missing) + ja: 312 keys (30 missing) + fr: 280 keys (62 missing) + ``` +4. Scan source code for **hard-coded strings** that should be extracted: + - String literals in JSX/TSX, template files, error messages. + - User-facing text in CLI output, log messages shown to users, email templates. + +### Phase 2a — String Extraction (Existing i18n Setup) + +Extract new or changed strings: + +1. Run the framework's extraction tool: + - `i18next-parser`: scans source for `t('key')` calls. + - `babel extract`: generates `.pot` files from Python source. + - `formatjs extract`: extracts from `intl.formatMessage()` calls. +2. Compare extracted keys against current base locale file. +3. For each **new key**: + - Verify the key name follows conventions. + - Add a description/comment for translator context. + - Set a default value in the base language. +4. For each **removed key** (no longer in source): + - Mark as deprecated (do not delete immediately — other branches may + reference it). + - Remove in a separate cleanup pass after merge. + +### Phase 2b — Bootstrap i18n (No Existing Setup) + +When the project has no i18n framework: + +1. Choose a framework appropriate to the tech stack. +2. Set up the base configuration (locale detection, fallback chain, + default namespace). +3. Create the directory structure: + ``` + src/i18n/ + config.ts # framework initialization + locales/ + en/ + common.json # shared keys + auth.json # feature-specific keys + zh-CN/ + common.json + auth.json + ``` +4. Replace hard-coded strings in source code with i18n function calls, + file by file. Prioritize: + - UI text visible to end users. + - Error messages and validation feedback. + - Email and notification templates. + - CLI output (if user-facing). + +### Phase 3 — Translation + +For each target language: + +1. Identify **missing keys** by diffing against the base locale. +2. Generate translations following these rules: + - Preserve **placeholders** exactly (`{name}`, `{{count}}`, `%s`). + - Respect **plural forms** for the target language. + - Maintain **HTML tags** and **Markdown** formatting if present. + - Keep translations **contextually appropriate** — do not translate + technical terms, brand names, or code identifiers. + - Respect **character limits** if specified (e.g., button labels). +3. For languages you cannot translate confidently, produce a draft and + flag it for human review: + ```json + { + "auth.mfa.prompt": { + "value": "请输入验证码", + "_review": true, + "_note": "Machine-translated; needs native review" + } + } + ``` + +### Phase 4 — Validation + +Run comprehensive checks: + +1. **Completeness**: every key in the base locale exists in all target locales. +2. **Placeholder integrity**: `{name}` in the base string must appear in every + translation — missing or extra placeholders cause runtime errors. +3. **Format string safety**: `%d`, `%s` counts must match between base and + translation. +4. **No untranslated base-language text**: detect translations that are identical + to the English base (may be untranslated). +5. **ICU syntax validity**: if using MessageFormat, parse each translation + string for syntax errors. +6. **Unused keys**: keys in locale files that no source code references + (wasted translator effort and bundle size). +7. **Key sorting**: ensure locale files are sorted alphabetically for + clean diffs. + +Produce a validation report: +``` +i18n Validation Report: + Base language: en (342 keys) + Languages: zh-CN, ja, fr + + zh-CN: 4 missing keys, 0 placeholder mismatches + Missing: auth.mfa.backup_codes_title, settings.theme.auto_label, ... + + ja: 30 missing keys, 2 placeholder mismatches + Mismatch: errors.rate_limit — base has {seconds}, ja translation missing + + fr: 62 missing keys, 0 placeholder mismatches + + Unused keys across all locales: 3 + legacy.old_feature_banner, onboarding.v1_welcome, ... +``` + +### Phase 5 — Write Back + +1. Write updated locale files with `file_write`. +2. Maintain consistent formatting: + - JSON: 2-space indent, sorted keys, trailing newline. + - YAML: 2-space indent, no document markers for single docs. + - PO/POT: standard gettext format. +3. If the project uses a key namespace pattern, place new keys in the + correct namespace file. + +## Error Handling + +| Situation | Action | +|---|---| +| Unknown i18n framework | Inspect imports and config files; if unidentifiable, ask the user. | +| Locale files use inconsistent formats (mix of JSON and YAML) | Standardize to the format used by the majority; migrate outliers. | +| Right-to-left language requested (Arabic, Hebrew) | Generate translations; flag that RTL CSS/layout support must be verified separately. | +| String contains complex ICU syntax (select, plural, nested) | Generate carefully; validate with a MessageFormat parser before writing. | + +## Limitations + +- Translation quality depends on language pair complexity. East Asian, + Arabic, and other structurally distant languages should be reviewed by + native speakers. +- This skill does not modify UI layout for text expansion (German text is + ~30% longer than English) — that is a design/CSS concern. +- Runtime locale detection and switching logic is framework-specific; + this skill sets up the data layer, not the runtime behavior. diff --git a/src/leapflow/skills/builtin_skills/project_analysis/SKILL.md b/src/leapflow/skills/builtin_skills/project_analysis/SKILL.md new file mode 100644 index 00000000..ce48ef20 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/project_analysis/SKILL.md @@ -0,0 +1,181 @@ +--- +name: project_analysis +description: "Project structure analysis, dependency audit, and tech stack identification" +version: 1.0.0 +metadata: + leapflow: + category: "development" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "development" + tags: ["project", "analysis", "dependencies", "architecture", "tech-stack"] + requires_tools: ["file_read", "file_list", "shell_run"] +platforms: [] +triggers: + - "analyze this project" + - "what does this project do" + - "project overview" + - "audit dependencies" + - "identify the tech stack" + - "project structure" + - "codebase analysis" + - "explain this repository" +--- + +# Project Analysis + +## Purpose + +Produce a comprehensive overview of a software project: its structure, tech +stack, dependencies, build system, and overall health. The output is a +structured report that helps someone unfamiliar with the project understand +its architecture and identify areas of concern. + +## Guiding Principles + +1. **Evidence-based** — Every claim about the project must be grounded in a + file or command output you actually observed. Do not infer a framework is + used unless you see its config or import. +2. **Proportional depth** — Spend more time on large or complex areas; do not + enumerate every file in a trivial directory. +3. **Actionable output** — The report should help the reader *do* something: + onboard faster, fix a vulnerability, upgrade a dependency. +4. **Non-destructive** — Never modify project files. All shell commands must + be read-only (e.g., `ls`, `cat`, `find`, `wc`, `grep`, package manager + list/audit commands). + +## Workflow + +### Phase 1 — Top-Level Scan + +Get the lay of the land: + +1. **List root directory** — `file_list` at the project root. Note key + marker files: + - `package.json` / `yarn.lock` / `pnpm-lock.yaml` → Node.js/JS/TS + - `pyproject.toml` / `setup.py` / `requirements.txt` / `uv.lock` → Python + - `go.mod` → Go + - `Cargo.toml` → Rust + - `pom.xml` / `build.gradle` → Java/Kotlin + - `Gemfile` → Ruby + - `Makefile`, `CMakeLists.txt`, `Dockerfile`, `docker-compose.yml` +2. **Read config files** — `file_read` on the primary manifest + (`package.json`, `pyproject.toml`, etc.) to extract: + - Project name, version, description + - Entry points / main modules + - Scripts / build commands +3. **Identify source layout** — conventional layouts (`src/`, `lib/`, `app/`, + `cmd/`, `internal/`, `pkg/`) vs flat structure. + +### Phase 2 — Tech Stack Identification + +From the evidence gathered, compile: + +| Layer | Technology | Evidence | +|---|---|---| +| Language(s) | e.g., Python 3.11 | `pyproject.toml` `requires-python` | +| Framework | e.g., FastAPI | import in `app/main.py` | +| Database | e.g., PostgreSQL | `DATABASE_URL` in `.env.example` | +| Build tool | e.g., uv | `uv.lock` present | +| CI/CD | e.g., GitHub Actions | `.github/workflows/` | +| Container | e.g., Docker | `Dockerfile` | +| Testing | e.g., pytest | `[tool.pytest]` in `pyproject.toml` | + +Do not guess — leave a cell blank if no evidence is found. + +### Phase 3 — Directory Structure Map + +Produce a tree-style outline of the major directories (depth ≤ 3), annotating +each with its purpose: + +``` +project-root/ +├── src/app/ # Application core (FastAPI routes, services) +├── src/models/ # Database models (SQLAlchemy) +├── tests/ # Test suite (pytest) +├── migrations/ # Alembic DB migrations +├── scripts/ # Dev/ops helper scripts +├── docs/ # Documentation +└── .github/workflows/ # CI pipelines +``` + +For each major directory, note: +- Approximate file count and dominant file types +- Key entry points or important files +- Any unusual or non-standard organization + +### Phase 4 — Dependency Audit + +Examine the project's declared dependencies: + +1. **Read the manifest** — extract all dependencies and their version + constraints. +2. **Categorize** — runtime vs dev-only vs optional. +3. **Flag concerns**: + - **Pinning**: Are versions pinned or floating? Wide ranges in production + dependencies are a risk. + - **Outdated**: If possible (e.g., `npm outdated`, `pip list --outdated`), + identify significantly outdated packages. + - **Known vulnerabilities**: If an audit command is available + (`npm audit`, `pip-audit`, `cargo audit`), run it. + - **Unused**: If the project has an unused-dependency detector, note any + findings. + - **License**: Flag any copyleft (GPL) dependencies in an otherwise + permissive-licensed project. +4. **Count**: Total dependencies, direct vs transitive (if lockfile present). + +### Phase 5 — Build & Run Assessment + +Determine how to build and run the project: + +1. Look for documented commands in `Makefile`, `package.json` scripts, + `pyproject.toml` scripts, or a `README`. +2. Check for required environment variables (`.env.example`, config files). +3. Note any non-obvious setup steps (database migration, code generation, + native compilation). +4. Assess whether a new contributor could get running with the documented + steps alone. + +### Phase 6 — Report + +Produce the final report: + +``` +## Project Overview +- **Name**: <name> +- **Description**: <1–2 sentences> +- **Primary Language**: <lang> +- **License**: <license> + +## Tech Stack +<table from Phase 2> + +## Directory Structure +<annotated tree from Phase 3> + +## Dependencies +- **Total**: N direct, M transitive +- **Health**: <summary> +- **Concerns**: <bulleted list of flagged issues> + +## Build & Run +- **Build command**: `<cmd>` +- **Run command**: `<cmd>` +- **Prerequisites**: <list> +- **Onboarding friction**: <low/medium/high with explanation> + +## Observations & Recommendations +1. <Actionable recommendation> +2. ... +``` + +## Error Handling + +| Situation | Action | +|---|---| +| No recognizable manifest file | State that the project type could not be identified; describe what was found at the root. | +| Cannot run audit commands (tool not installed) | Skip the vulnerability check; note it was not performed and suggest the user run it manually. | +| Very large monorepo | Focus on the top-level structure and the most active or largest sub-packages; do not attempt to map everything. | +| Binary or generated files dominate | Note this and focus analysis on the source directories. | diff --git a/src/leapflow/skills/builtin_skills/security_audit/SKILL.md b/src/leapflow/skills/builtin_skills/security_audit/SKILL.md new file mode 100644 index 00000000..8326f359 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/security_audit/SKILL.md @@ -0,0 +1,200 @@ +--- +name: security_audit +description: "Security vulnerability scanning strategy, threat modeling, and fix recommendations" +version: 1.0.0 +metadata: + leapflow: + category: "security" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "security" + tags: ["security", "vulnerability", "audit", "OWASP", "threat-model", "hardening"] + requires_tools: ["file_read", "shell_run"] +platforms: [] +triggers: + - "security audit" + - "vulnerability scan" + - "安全审计" + - "漏洞扫描" + - "security check" + - "threat model" + - "harden this code" + - "OWASP check" +--- + +# Security Audit + +## Purpose + +Conduct a systematic security audit of a codebase or configuration, identifying +vulnerabilities, assessing their severity, and providing actionable remediation +guidance. This skill follows a structured methodology inspired by OWASP and +industry threat-modeling frameworks — not a checklist run, but a risk-prioritized +analysis that focuses effort where exploitability is highest. + +## Guiding Principles + +1. **Attacker's perspective** — Think about how each finding could be exploited, + not just whether a pattern looks suspicious. A theoretical vulnerability + with no practical attack path is low priority. +2. **Risk = Likelihood × Impact** — Prioritize findings by both exploitability + and damage potential. A SQL injection in a public endpoint outranks a + minor info leak in an admin-only debug page. +3. **Evidence-based findings** — Every vulnerability report includes the exact + code location, a concrete attack scenario, and a specific remediation. + No vague warnings. +4. **Defense in depth** — Do not stop after finding one vulnerability. Assess + whether multiple layers of defense exist and where they are weakest. +5. **Fix the root cause** — Recommend fixes that address the underlying design + flaw, not just the specific instance. + +## Workflow + +### Phase 1 — Scope and Reconnaissance + +1. Map the **attack surface**: + - Entry points: HTTP endpoints, CLI commands, message queues, file uploads, + IPC sockets, deserialization points. + - Authentication boundaries: which paths are public, authenticated, or + require elevated privileges. + - Data flows: where sensitive data (credentials, PII, tokens) enters, + is processed, stored, and exits. +2. Identify the **technology stack** and its known vulnerability patterns: + - Language-specific: Python (pickle, eval, SSTI), JS (prototype pollution, + ReDoS), Java (deserialization, XXE), Go (integer overflow, goroutine leak). + - Framework-specific: check for known CVEs in the framework version. + - Dependency-specific: run `shell_run` with `pip audit`, `npm audit`, + `cargo audit`, or `govulncheck` as appropriate. +3. Review **security configuration**: + - TLS settings, CORS policy, CSP headers, cookie flags. + - Secret management: how are API keys, DB passwords, and tokens stored? + - Logging: are sensitive values redacted? + +### Phase 2 — Automated Scanning + +Run available automated tools: + +| Tool | Scope | Command | +|---|---|---| +| `pip audit` / `npm audit` | Dependency CVEs | `pip audit --format json` | +| `bandit` (Python) | Source code patterns | `bandit -r src/ -f json` | +| `semgrep` | Multi-language patterns | `semgrep --config auto src/` | +| `trivy fs` | Filesystem + deps | `trivy fs --severity HIGH,CRITICAL .` | +| `gitleaks` | Secrets in history | `gitleaks detect --source .` | + +Parse results and de-duplicate. Automated findings are leads, not conclusions +— each must be validated manually in Phase 3. + +### Phase 3 — Manual Analysis (OWASP Top 10 Focus) + +Systematically examine code for each OWASP category: + +**A01 — Broken Access Control**: +- Are authorization checks enforced at the handler level, not just the router? +- Can a user access another user's resources by changing IDs (IDOR)? +- Are admin endpoints protected by role checks, not just authentication? + +**A02 — Cryptographic Failures**: +- Are passwords hashed with bcrypt/scrypt/argon2 (not MD5/SHA1)? +- Are secrets stored in environment variables or vaults (not code/config files)? +- Is data in transit encrypted (TLS 1.2+)? Data at rest? + +**A03 — Injection**: +- SQL: parameterized queries everywhere? No string concatenation in queries? +- Command: is `subprocess` called with `shell=False`? Is user input sanitized? +- Template: is user input escaped before rendering in templates? +- XSS: is output encoding applied in all HTML contexts? + +**A04 — Insecure Design**: +- Are rate limits enforced on authentication endpoints? +- Is there account lockout or CAPTCHA after repeated failures? +- Are business logic constraints enforced server-side (not just client)? + +**A05 — Security Misconfiguration**: +- Are debug modes, default credentials, or verbose error pages exposed? +- Are unnecessary services, ports, or features enabled? +- Are HTTP security headers set (X-Frame-Options, X-Content-Type-Options)? + +**A06 — Vulnerable Components**: +- Cross-reference dependency scan results with NVD/OSV databases. +- Check for unmaintained dependencies (no commits in >2 years). + +**A07 — Authentication Failures**: +- Are sessions invalidated on logout and password change? +- Are tokens short-lived with proper refresh mechanisms? +- Is MFA supported for sensitive operations? + +**A08 — Data Integrity Failures**: +- Are software updates and CI/CD pipelines integrity-verified? +- Is deserialization of untrusted data avoided or validated? + +**A09 — Logging and Monitoring**: +- Are authentication events (login, failure, lockout) logged? +- Are logs protected from injection and tampering? +- Is there alerting on anomalous patterns? + +**A10 — SSRF**: +- Are outbound requests validated against an allowlist? +- Can user input influence internal URLs or DNS resolution? + +### Phase 4 — Severity Assessment + +Rate each confirmed finding using CVSS-like scoring: + +| Severity | Criteria | Example | +|---|---|---| +| **Critical** | Remote exploitation, no auth required, data breach likely | Unauthenticated SQL injection on public endpoint | +| **High** | Exploitation requires low-privilege auth, significant impact | IDOR allowing access to other users' data | +| **Medium** | Requires specific conditions, moderate impact | XSS in admin panel requiring social engineering | +| **Low** | Theoretical risk, minimal real-world impact | Information disclosure of framework version | +| **Info** | Best-practice deviation, no direct risk | Missing security header on non-sensitive endpoint | + +### Phase 5 — Report + +Produce a structured security audit report: + +``` +## Executive Summary +<Overall risk posture: Critical/High/Medium/Low> +<N critical, N high, N medium, N low findings> + +## Critical Findings +### [CRIT-001] <Title> +- **Location**: <file:line> +- **Description**: <what the vulnerability is> +- **Attack Scenario**: <step-by-step exploitation> +- **Impact**: <what an attacker gains> +- **Remediation**: <specific code change with example> +- **References**: <CWE/CVE/OWASP link> + +## High Findings +### [HIGH-001] ... + +## Medium / Low / Info Findings +... + +## Positive Security Practices +<Things the project does well — reinforcement matters> + +## Recommended Hardening Steps +1. <Prioritized action items> +``` + +## Error Handling + +| Situation | Action | +|---|---| +| Scanning tools not installed | Report which tools are missing; provide install commands; proceed with manual analysis. | +| Codebase too large for full audit | Focus on the highest-risk areas: authentication, input processing, external interfaces. | +| Encrypted or obfuscated code | Report as a limitation; audit only readable portions. | +| Finding severity is ambiguous | Default to the higher severity and note the uncertainty. | + +## Limitations + +- This skill performs static analysis and code review; it does not execute + dynamic tests (penetration testing, fuzzing) or interact with running services. +- Automated tool results depend on tool availability in the environment. +- Findings are based on code as read; runtime behavior may differ due to + configuration, middleware, or infrastructure layers not visible in source. diff --git a/src/leapflow/skills/builtin_skills/shell_automation/SKILL.md b/src/leapflow/skills/builtin_skills/shell_automation/SKILL.md new file mode 100644 index 00000000..47f6ad2b --- /dev/null +++ b/src/leapflow/skills/builtin_skills/shell_automation/SKILL.md @@ -0,0 +1,166 @@ +--- +name: shell_automation +description: "Natural language to shell commands with safety review and step-by-step execution" +version: 1.0.0 +metadata: + leapflow: + category: "automation" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "automation" + tags: ["shell", "terminal", "automation", "commands", "scripting"] + requires_tools: ["shell_run"] +platforms: [] +triggers: + - "run a shell command" + - "automate this task" + - "write a script to" + - "shell automation" + - "batch process" + - "execute commands" + - "terminal commands for" + - "help me with the command line" +--- + +# Shell Automation + +## Purpose + +Translate a user's natural-language goal into a safe, step-by-step shell +command plan. Each command is reviewed for safety before execution, and +intermediate results are verified before proceeding to the next step. The +skill prioritizes safety and predictability over speed. + +## Guiding Principles + +1. **Safety first** — Every command is classified by risk level before + execution. Destructive or irreversible operations require explicit + acknowledgement of consequences. +2. **Incremental execution** — Run one logical step at a time. Verify the + output before proceeding. Never chain destructive commands with `&&`. +3. **Least privilege** — Use the minimum permissions necessary. Avoid `sudo` + unless the user explicitly requests it and the task genuinely requires it. +4. **Idempotent preference** — Prefer commands that are safe to re-run + (e.g., `mkdir -p` over `mkdir`, `cp` with backup over `mv`). +5. **Transparency** — Show every command before running it. Explain what it + does and what side effects it has. No hidden operations. + +## Workflow + +### Phase 1 — Understand the Goal + +Parse the user's request into: + +- **Objective**: What end state does the user want? +- **Scope**: Which files, directories, or services are involved? +- **Constraints**: OS, shell flavor (bash/zsh/fish), available tools, + environment (local dev, CI, production server). +- **Risk tolerance**: Is the user experimenting or operating on production + data? + +If anything is ambiguous, ask a clarifying question before planning. + +### Phase 2 — Plan the Command Sequence + +Design an ordered list of commands. For each command: + +1. **Write the command** — use proper flags and quoting. +2. **Explain it** — one sentence on what it does. +3. **Classify the risk**: + +| Risk Level | Criteria | Examples | +|---|---|---| +| **Safe** | Read-only, no side effects | `ls`, `cat`, `grep`, `find`, `wc`, `df`, `ps` | +| **Low** | Creates new files/dirs, non-destructive writes | `mkdir -p`, `touch`, `tee`, `cp` (no overwrite) | +| **Medium** | Modifies existing files, installs packages, changes config | `sed -i`, `pip install`, `chmod`, `git commit` | +| **High** | Deletes data, stops services, modifies system state | `rm`, `kill`, `systemctl stop`, `drop table` | +| **Critical** | Irreversible, wide blast radius | `rm -rf /`, `dd`, `mkfs`, `git push --force` | + +4. **Add a verification step** after each non-trivial command — a read-only + command that confirms the expected outcome (e.g., `ls` after `mv`, + `cat` after `sed`). + +Present the full plan before executing anything. + +### Phase 3 — Safety Review + +Before execution, perform a checklist: + +- [ ] No command uses `sudo` unless explicitly justified. +- [ ] No `rm` without a narrow, explicit target (never `rm -rf` with a + variable or glob that could expand dangerously). +- [ ] All file paths are absolute or explicitly scoped to the working + directory. +- [ ] No secrets, passwords, or tokens appear in plaintext in any command. +- [ ] Pipes and redirects do not silently overwrite important files (prefer + `>>` over `>` when appending; use `tee` for visibility). +- [ ] The plan has a **rollback path** for any medium-or-higher risk step + (e.g., "if this fails, run X to restore state"). + +If a Critical-risk command is part of the plan, add a prominent warning block: + +``` +⚠️ CRITICAL: The following command is irreversible. + Command: <cmd> + Effect: <what it destroys/changes> + Verify: <how to confirm this is correct before running> +``` + +### Phase 4 — Step-by-Step Execution + +Execute the plan one command at a time: + +1. **Show** the command and its explanation. +2. **Run** it via `shell_run`. +3. **Check** the exit code and output: + - Exit 0 + expected output → proceed to the next step. + - Non-zero exit or unexpected output → stop, diagnose, and report. +4. **Run the verification step** if one was planned. +5. **Log** a brief result: "✓ Created directory `/tmp/backup`" or + "✗ Failed: permission denied on `/etc/hosts`". + +Do NOT proceed past a failed step unless: +- The failure is explicitly expected (e.g., `grep` returning 1 for no match). +- The user confirms they want to skip and continue. + +### Phase 5 — Summary + +After all steps complete (or after a halt): + +``` +## Execution Summary +- Steps completed: N / M +- Status: <all succeeded / stopped at step K> + +## Commands Executed +1. `<cmd>` — ✓ <result> +2. `<cmd>` — ✗ <error> + +## Next Steps +- <Anything the user should do manually or verify> +``` + +## Error Handling + +| Situation | Action | +|---|---| +| Command not found | Check if the tool is installed (`which <cmd>`); suggest installation if missing. | +| Permission denied | Do NOT auto-escalate to `sudo`. Report the error and let the user decide. | +| Command hangs (timeout) | Report the timeout; suggest running with `timeout <N>s` wrapper or checking for interactive prompts. | +| Ambiguous user request | Ask for clarification rather than guessing — a wrong guess with `rm` is unrecoverable. | +| User requests a dangerous pattern (`rm -rf *`, `chmod 777`) | Explain the risk clearly; suggest a safer alternative; proceed only if the user explicitly confirms after understanding the consequences. | + +## Forbidden Patterns + +The following patterns must NEVER be executed without explicit, informed user +confirmation and a clear justification: + +- `rm -rf /` or `rm -rf ~` or any `rm` targeting a root-level directory +- `:(){ :|:& };:` or any fork bomb variant +- `dd` writing to a block device +- `chmod -R 777` on system directories +- `> /dev/sda` or equivalent destructive redirects +- Any command that downloads and pipes directly to `sh`/`bash` without + the user reviewing the script first (e.g., `curl ... | sh`) diff --git a/src/leapflow/skills/builtin_skills/test_writer/SKILL.md b/src/leapflow/skills/builtin_skills/test_writer/SKILL.md new file mode 100644 index 00000000..ed3b0709 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/test_writer/SKILL.md @@ -0,0 +1,195 @@ +--- +name: test_writer +description: "Generate unit and integration tests from source code with edge-case coverage" +version: 1.0.0 +metadata: + leapflow: + category: "development" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "development" + tags: ["testing", "unit-test", "integration-test", "test-generation", "coverage"] + requires_tools: ["file_read", "file_write", "shell_run"] +platforms: [] +triggers: + - "write tests" + - "generate tests" + - "create test cases" + - "unit test" + - "写测试" + - "生成测试用例" + - "add test coverage" + - "integration test" +--- + +# Test Writer + +## Purpose + +Generate high-quality test cases — unit, integration, and boundary tests — by +analyzing source code, identifying testable units, determining edge cases, and +producing test code that follows the project's existing conventions. Tests must +not only pass but must catch real regressions. + +## Guiding Principles + +1. **Tests document behavior** — A test file is the executable specification of + what the code does. Name tests after the behavior they verify, not the + implementation they exercise. +2. **Arrange-Act-Assert** — Every test has three clear sections: set up state, + perform the action, verify the outcome. One action per test. +3. **Edge cases over happy paths** — Happy paths are obvious; value comes from + testing boundaries, error paths, empty inputs, and concurrent access. +4. **Independence** — Tests must not depend on each other's execution order or + shared mutable state. Each test sets up and tears down its own context. +5. **Follow the project** — Use the project's existing test framework, directory + layout, naming conventions, and fixture patterns. Do not introduce new + testing libraries without explicit approval. + +## Workflow + +### Phase 1 — Analyze the Target Code + +1. Read the source file(s) with `file_read`. +2. Identify: + - **Public API surface**: functions, methods, and classes intended for + external use. + - **Input types and constraints**: what each parameter accepts, valid + ranges, required vs optional. + - **Output types**: return values, raised exceptions, side effects + (file writes, network calls, state mutations). + - **Dependencies**: external services, databases, file system, time, + randomness — these will need mocking or stubbing. + - **Existing tests**: check if a test file already exists; understand + what is already covered. + +3. Determine the test framework by examining existing tests: + - Python: `pytest`, `unittest` + - JavaScript/TypeScript: `jest`, `vitest`, `mocha` + - Go: standard `testing` package + - Other: detect from import patterns or config files. + +### Phase 2 — Design Test Cases + +For each testable unit, enumerate cases across these categories: + +**Normal behavior**: +- Typical valid inputs producing expected outputs. +- Multiple valid input combinations if the function is polymorphic. + +**Boundary conditions**: +- Empty inputs (empty string, empty list, zero, None/null). +- Minimum and maximum valid values. +- Single-element collections. +- Boundary of numeric ranges (off-by-one). + +**Error paths**: +- Invalid inputs that should raise exceptions or return error codes. +- Missing required parameters. +- Type mismatches (if the language is dynamically typed). + +**State transitions** (for stateful code): +- Initial state → action → expected state. +- Invalid state transitions that should be rejected. + +**Integration points** (for integration tests): +- Correct interaction with mocked dependencies. +- Behavior when dependencies fail (timeout, error, empty response). + +Produce a test plan: +``` +Target: calculate_discount(price, tier, coupon_code) +Cases: + 1. Normal: valid price + gold tier → 20% discount + 2. Normal: valid price + no tier → 0% discount + 3. Boundary: price = 0 → discount = 0 + 4. Boundary: price = MAX_FLOAT → no overflow + 5. Error: negative price → raises ValueError + 6. Error: unknown tier → raises ValueError + 7. Error: expired coupon → returns original price + warning + 8. Integration: coupon service unreachable → graceful fallback +``` + +### Phase 3 — Generate Test Code + +Write test code following these rules: + +1. **File location**: place tests where the project expects them (e.g., + `tests/test_<module>.py`, `__tests__/<module>.test.ts`, `<module>_test.go`). +2. **Test naming**: `test_<function>_<scenario>_<expected_outcome>`. + Examples: `test_calculate_discount_negative_price_raises_value_error`, + `test_parse_config_empty_file_returns_defaults`. +3. **Fixtures and setup**: extract common setup into fixtures (`@pytest.fixture`, + `beforeEach`, `TestMain`). Keep fixtures close to the tests that use them. +4. **Mocking strategy**: + - Mock at the boundary: external services, I/O, time, randomness. + - Do not mock the unit under test or its core logic. + - Use dependency injection where the code supports it; patch as last resort. +5. **Assertions**: + - Assert specific values, not just truthiness. + - For exceptions: assert both the exception type and message content. + - For collections: assert length and key elements, not just non-empty. +6. **Readability**: each test should be understandable in isolation without + reading other tests. Inline small data; use descriptive variable names. + +### Phase 4 — Verify Tests + +After writing tests: + +1. Run the test suite with `shell_run`: + - Python: `python -m pytest <test_file> -v` + - JS/TS: `npx jest <test_file>` or `npx vitest run <test_file>` + - Go: `go test -v -run <TestName> ./<package>` +2. Verify all tests **pass**. If any fail: + - Read the failure message carefully. + - Distinguish between a bug in the test (wrong expectation) and a bug + in the source code (genuine regression). + - Fix test bugs; report source code bugs to the user. +3. Check that tests **fail when they should**: temporarily break the source + logic and confirm the test catches it (mutation testing principle). +4. Review test output for: + - Flaky behavior (tests that pass/fail non-deterministically). + - Slow tests (>1 second for unit tests indicates I/O leaking in). + - Missing coverage for the identified edge cases. + +### Phase 5 — Report + +Summarize the test generation: + +``` +## Test Summary +- File: tests/test_discount.py +- Tests added: 8 (5 unit, 2 boundary, 1 integration) +- All passing: Yes +- Mocks used: coupon_service (httpx response stub) +- Coverage delta: +12% for discount.py (estimated) + +## Notable Edge Cases Covered +1. Negative price rejection +2. Coupon service timeout fallback +3. Floating-point precision at MAX_FLOAT + +## Gaps / Recommendations +- Concurrent discount calculations not tested (requires async fixtures) +- Property-based testing recommended for numeric inputs (hypothesis/fast-check) +``` + +## Error Handling + +| Situation | Action | +|---|---| +| No existing test framework detected | Ask the user which framework to use; default to the language's standard (`pytest`, `jest`, `go test`). | +| Source code has no clear testable units (monolithic function) | Suggest refactoring; write tests for observable inputs/outputs of the monolith. | +| Tests pass but are tautological (assert True) | Flag as a quality issue; rewrite with meaningful assertions. | +| Cannot run tests (missing dependencies) | Generate the test code and provide the exact install/run commands the user needs. | + +## Limitations + +- Test generation requires access to the source code and an understanding of + the project's test infrastructure. +- Integration tests that require live services (databases, APIs) will use mocks; + the user must configure real service connections for end-to-end testing. +- Code coverage measurement requires project-specific tooling configuration + that this skill does not modify. diff --git a/src/leapflow/skills/builtin_skills/web_research/SKILL.md b/src/leapflow/skills/builtin_skills/web_research/SKILL.md new file mode 100644 index 00000000..dafba736 --- /dev/null +++ b/src/leapflow/skills/builtin_skills/web_research/SKILL.md @@ -0,0 +1,141 @@ +--- +name: web_research +description: "Multi-step web search with cross-verification and structured synthesis" +version: 1.0.0 +metadata: + leapflow: + category: "research" + source: "builtin" + confidence: 1.0 + quality_score: 1.0 + hermes: + category: "research" + tags: ["web", "search", "research", "synthesis", "fact-checking"] + requires_tools: ["web_search", "web_fetch"] +platforms: [] +triggers: + - "research this topic" + - "search the web for" + - "find information about" + - "look up" + - "investigate online" + - "web research" + - "deep search" + - "find and summarize" +--- + +# Web Research + +## Purpose + +Conduct rigorous, multi-step web research that goes beyond a single search query. +This skill orchestrates a structured research workflow: planning a search strategy, +executing multiple targeted queries, cross-verifying claims across independent +sources, and synthesizing findings into a coherent, well-sourced report. + +## Guiding Principles + +1. **Breadth before depth** — Cast a wide net with diverse queries before drilling + into any single source. Reformulate queries when initial results are thin. +2. **Source triangulation** — Never trust a single source. Every key claim must + appear in at least two independent sources before it is treated as established. +3. **Recency awareness** — Prefer recent sources for fast-moving topics. Flag + when the most recent source is older than the topic warrants. +4. **Bias detection** — Note when sources share a common owner, funding body, or + obvious editorial slant. Weight accordingly. +5. **Transparency** — Every factual statement in the final report must link back + to the source URL it was derived from. + +## Workflow + +### Phase 1 — Understand the Question + +Before any search, decompose the user's request: + +- Identify the **core question** (what must be answered). +- List **sub-questions** that feed into the core answer. +- Note any **constraints**: date range, geography, domain, language. +- Decide on a **search budget**: how many queries are proportionate to the task + (typically 3–8 queries; simple factual lookups need fewer). + +Output a brief research plan (≤ 6 bullet points) and confirm with yourself +before proceeding. + +### Phase 2 — Execute Searches + +For each sub-question, craft a targeted search query: + +1. Use **specific, keyword-rich** queries — avoid full sentences. +2. Vary phrasing across queries to surface different result sets. +3. Include **domain-specific qualifiers** when useful + (e.g., `site:arxiv.org`, `filetype:pdf`, `"exact phrase"`). +4. After each `web_search` call, scan the snippets. If a result looks + authoritative or contains the needed data, fetch the full page with + `web_fetch`. +5. Record each source: URL, title, date (if available), and a 1–2 sentence + summary of what it contributes. + +Do NOT fetch every result — only those that plausibly advance an answer. + +### Phase 3 — Cross-Verify + +Lay out the key claims you have gathered and check: + +- Does each claim appear in **≥ 2 independent sources**? +- Are any claims **contradicted** by another source? If so, note the + contradiction and assess which source is more credible (recency, authority, + methodology). +- Are there **gaps** — sub-questions that no source adequately addresses? + If so, run 1–2 additional targeted searches. + +### Phase 4 — Synthesize + +Produce a structured report: + +``` +## Summary +<2–4 sentence executive summary> + +## Key Findings +1. <Finding with inline source reference [1]> +2. ... + +## Contradictions / Uncertainties +- <Any unresolved conflicts or low-confidence claims> + +## Sources +[1] <title> — <url> (accessed <date>) +[2] ... +``` + +Rules for the report: +- Lead with the answer — do not bury it after methodology. +- Use numbered source references; every factual claim must cite at least one. +- Clearly separate established facts from speculation or single-source claims. +- If the research is inconclusive, say so explicitly rather than hedging. + +### Phase 5 — Self-Check + +Before delivering: + +- Re-read the original question. Does the report actually answer it? +- Are all source URLs valid (no hallucinated links)? +- Is the report length proportionate to the question's complexity? + +## Error Handling + +| Situation | Action | +|---|---| +| `web_search` returns no results | Reformulate the query with broader or alternative terms; try up to 2 reformulations before reporting the gap. | +| `web_fetch` fails or returns empty | Note the URL as inaccessible; do not cite it. Rely on the search snippet if it contained the needed fact. | +| Contradictory sources with equal credibility | Present both perspectives explicitly; do not silently pick one. | +| Topic is too broad | Ask the user to narrow scope before consuming the full search budget. | + +## Limitations + +- This skill relies on `web_search` and `web_fetch` tools. If they are + unavailable, the skill cannot execute. +- Results are bounded by what the search engine indexes; paywalled or + dynamically rendered content may be inaccessible. +- Always note when a topic requires expertise that web search alone cannot + reliably provide (medical, legal, financial advice). diff --git a/src/leapflow/skills/curator.py b/src/leapflow/skills/curator.py index 561622ba..8f380d21 100644 --- a/src/leapflow/skills/curator.py +++ b/src/leapflow/skills/curator.py @@ -12,14 +12,43 @@ from __future__ import annotations import enum +import json import logging +import re import time from dataclasses import dataclass, field -from typing import Any, Dict, Optional, Protocol, runtime_checkable +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Protocol, runtime_checkable + +if TYPE_CHECKING: + from leapflow.llm.base import LLMProvider logger = logging.getLogger(__name__) +# ── Consolidation types ── + + +class ConsolidationAction(str, enum.Enum): + """Possible actions when two skills overlap significantly.""" + + MERGE = "merge" # Combine into a single skill + KEEP_SEPARATE = "keep_separate" # Intentional overlap, keep both + DEPRECATE_ONE = "deprecate_one" # One skill supersedes the other + + +@dataclass(frozen=True) +class ConsolidationSuggestion: + """LLM-generated suggestion for consolidating overlapping skills.""" + + skill_a: str + skill_b: str + action: ConsolidationAction + reason: str + confidence: float + merged_name: str = "" + merged_description: str = "" + + # ── Curation state machine ── class CurationState(str, enum.Enum): @@ -104,17 +133,23 @@ def __init__( store: SkillCurationStore, *, event_bus: Optional[Any] = None, + llm_provider: Optional[LLMProvider] = None, stale_after_days: int = _DEFAULT_STALE_DAYS, archive_after_days: int = _DEFAULT_ARCHIVE_DAYS, ) -> None: self._store = store self._event_bus = event_bus + self._llm_provider: Optional[LLMProvider] = llm_provider self._stale_after_days = stale_after_days self._archive_after_days = archive_after_days self._last_sweep_time: float = 0.0 # In-memory cache for fast lookups (lazily populated) self._cache: Optional[Dict[str, SkillCurationEntry]] = None + def set_llm(self, provider: LLMProvider) -> None: + """Inject or replace the LLM provider (back-reference pattern).""" + self._llm_provider = provider + # ── Cache management ── def _ensure_cache(self) -> Dict[str, SkillCurationEntry]: @@ -319,6 +354,178 @@ def get_curation_report(self) -> CurationReport: """Generate a current-state curation report.""" return self._build_report([]) + # ── LLM-powered consolidation ── + + async def consolidate( + self, + skill_entries: List[Any], + *, + llm_provider: Optional[LLMProvider] = None, + ) -> List[ConsolidationSuggestion]: + """Detect overlapping skills and suggest merges via LLM. + + Collects active skills from *skill_entries* (SkillEntry instances), + groups them by category, and asks the LLM to identify significant + overlaps. Returns a list of :class:`ConsolidationSuggestion` + instances for user review — no automatic mutations are performed. + + Args: + skill_entries: Active SkillEntry instances from SkillIndex. + llm_provider: Override the instance-level LLM provider for + this single call. + + Returns: + Suggestions list (empty when no LLM is available or no + overlaps are detected). + """ + provider = llm_provider or self._llm_provider + if provider is None: + logger.warning( + "curator.consolidate: no LLM provider available; " + "returning empty suggestions" + ) + return [] + + # Group skills by category for efficient comparison + by_category: Dict[str, List[Any]] = {} + for entry in skill_entries: + cat = getattr(entry, "category", "") or "uncategorized" + by_category.setdefault(cat, []).append(entry) + + suggestions: List[ConsolidationSuggestion] = [] + + for category, group in by_category.items(): + if len(group) < 2: + continue + prompt = self._build_consolidation_prompt(category, group) + try: + response = await provider.achat( + [ + { + "role": "system", + "content": ( + "You are a skill-catalog analyst. " + "Respond ONLY with the JSON array described " + "in the user message. No markdown fences, " + "no commentary." + ), + }, + {"role": "user", "content": prompt}, + ], + stream=False, + ) + parsed = self._parse_consolidation_response(response.content) + suggestions.extend(parsed) + except Exception as exc: + logger.warning( + "curator.consolidate: LLM call failed for " + "category=%s error=%s", + category, exc, + ) + + logger.info( + "curator.consolidate: %d suggestion(s) across %d categories", + len(suggestions), len(by_category), + ) + return suggestions + + @staticmethod + def _build_consolidation_prompt( + category: str, entries: List[Any], + ) -> str: + """Format skill entries into a structured LLM prompt.""" + skill_lines: List[str] = [] + for entry in entries: + name = getattr(entry, "name", "unknown") + desc = getattr(entry, "description", "")[:200] + tags = ", ".join(getattr(entry, "tags", ()) or ()) + triggers = ", ".join(getattr(entry, "triggers", ()) or ()) + parts = [f" name: {name}", f" description: {desc}"] + if tags: + parts.append(f" tags: {tags}") + if triggers: + parts.append(f" triggers: {triggers}") + skill_lines.append("\n".join(parts)) + + skills_block = "\n---\n".join(skill_lines) + + return ( + f"Category: {category}\n" + f"Skills ({len(entries)}):\n" + f"{skills_block}\n\n" + "Analyze these skills for significant overlap. " + "For each overlapping pair, produce a JSON object with:\n" + ' "skill_a": <name>,\n' + ' "skill_b": <name>,\n' + ' "action": "merge" | "keep_separate" | "deprecate_one",\n' + ' "reason": <concise explanation>,\n' + ' "confidence": <0.0-1.0>,\n' + ' "merged_name": <suggested name if merge>,\n' + ' "merged_description": <suggested description if merge>\n\n' + "Return a JSON array of these objects. " + "If no overlaps exist, return [].\n" + "Do NOT wrap in markdown code fences." + ) + + @staticmethod + def _parse_consolidation_response( + raw: str, + ) -> List[ConsolidationSuggestion]: + """Defensively parse LLM JSON into ConsolidationSuggestion list.""" + # Strip markdown code fences if present + text = raw.strip() + fence_match = re.search(r"```(?:json)?\s*\n?(.*?)```", text, re.DOTALL) + if fence_match: + text = fence_match.group(1).strip() + + try: + data = json.loads(text) + except json.JSONDecodeError as exc: + logger.warning( + "curator.consolidate: malformed JSON from LLM: %s", exc + ) + return [] + + if not isinstance(data, list): + logger.warning( + "curator.consolidate: expected JSON array, got %s", + type(data).__name__, + ) + return [] + + action_map = {a.value: a for a in ConsolidationAction} + suggestions: List[ConsolidationSuggestion] = [] + + for item in data: + if not isinstance(item, dict): + continue + try: + action_str = str(item.get("action", "")).lower() + action = action_map.get(action_str) + if action is None: + logger.debug( + "curator.consolidate: unknown action '%s', skipping", + action_str, + ) + continue + suggestion = ConsolidationSuggestion( + skill_a=str(item.get("skill_a", "")), + skill_b=str(item.get("skill_b", "")), + action=action, + reason=str(item.get("reason", "")), + confidence=float(item.get("confidence", 0.5)), + merged_name=str(item.get("merged_name", "")), + merged_description=str(item.get("merged_description", "")), + ) + suggestions.append(suggestion) + except (ValueError, TypeError) as exc: + logger.debug( + "curator.consolidate: skipping malformed entry: %s", exc + ) + continue + + return suggestions + # ── Internal helpers ── def _build_report(self, transitions: list[CurationTransition]) -> CurationReport: @@ -363,6 +570,8 @@ def _emit_transition( __all__ = [ + "ConsolidationAction", + "ConsolidationSuggestion", "CurationState", "CurationReport", "CurationTransition", diff --git a/src/leapflow/skills/discovery.py b/src/leapflow/skills/discovery.py index 54d9a1a0..29b40165 100644 --- a/src/leapflow/skills/discovery.py +++ b/src/leapflow/skills/discovery.py @@ -1,9 +1,11 @@ # Copyright (c) Alibaba, Inc. and its affiliates. """Skill discovery tools — exposed to LLM for progressive disclosure. -Provides two tool handlers registered into the unified tool system: -- skills_list: Layer 1 — compact metadata listing with optional keyword filter -- skill_view: Layer 2 — full SKILL.md content for a specific skill +Provides four tool handlers registered into the unified tool system: +- skills_list: Layer 1 — compact metadata listing with optional keyword filter +- skill_view: Layer 2 — full SKILL.md content for a specific skill +- skill_list_files: Layer 3a — list support files in a skill directory +- skill_file_read: Layer 3b — read individual support file from a skill directory Module-level configuration pattern: call configure() at startup to inject SkillIndex and SkillInjector instances without coupling to DI framework. @@ -11,7 +13,9 @@ from __future__ import annotations import logging -from typing import Any, Dict, Optional +import os +from pathlib import Path, PurePosixPath +from typing import Any, Dict, List, Optional logger = logging.getLogger(__name__) @@ -151,3 +155,108 @@ async def skill_view(params: Dict[str, Any]) -> Dict[str, Any]: result["total_chars"] = len(content) return result + + +def _validate_relative_path(file_path: str) -> Optional[str]: + """Validate that *file_path* is relative and does not escape the skill dir. + + Returns an error message string if invalid, or ``None`` if safe. + """ + if not file_path: + return "file_path is required" + # Reject absolute paths (Unix and Windows) + if file_path.startswith("/") or file_path.startswith("\\"): + return "Absolute paths are not allowed" + # Reject any component that walks upward + parts = PurePosixPath(file_path).parts + if ".." in parts: + return "Path traversal ('..') is not allowed" + return None + + +async def skill_list_files(params: Dict[str, Any]) -> Dict[str, Any]: + """Layer 3a: List support files available in a skill directory. + + Params: + name (str, required): Skill name to look up. + + Returns dict with ok, name, files[] fields. + """ + if _skill_injector is None: + return {"ok": False, "error": "Skill injector not initialized"} + + name = str(params.get("name", "")).strip() + if not name: + return {"ok": False, "error": "Skill name required"} + + skill_dir = _skill_injector.find_skill_dir(name) + if skill_dir is None: + return {"ok": False, "error": f"Skill '{name}' not found"} + + files: List[str] = [] + skill_path = Path(skill_dir) + for root, _dirs, filenames in os.walk(skill_path): + root_path = Path(root) + for fn in sorted(filenames): + rel = (root_path / fn).relative_to(skill_path) + files.append(str(rel)) + + return {"ok": True, "name": name, "files": sorted(files), "path": str(skill_dir)} + + +async def skill_file_read(params: Dict[str, Any]) -> Dict[str, Any]: + """Layer 3b: Read an individual support file from a skill directory. + + Params: + name (str, required): Skill name to look up. + file_path (str, required): Relative path within the skill directory. + + Returns dict with ok, name, file_path, content, path fields. + """ + if _skill_injector is None: + return {"ok": False, "error": "Skill injector not initialized"} + + name = str(params.get("name", "")).strip() + if not name: + return {"ok": False, "error": "Skill name required"} + + file_path = str(params.get("file_path", "")).strip() + path_err = _validate_relative_path(file_path) + if path_err: + return {"ok": False, "error": path_err} + + skill_dir = _skill_injector.find_skill_dir(name) + if skill_dir is None: + return {"ok": False, "error": f"Skill '{name}' not found"} + + skill_root = Path(skill_dir).resolve() + target = (skill_root / file_path).resolve() + try: + target.relative_to(skill_root) + except ValueError: + return {"ok": False, "error": "Path escapes skill directory"} + + if not target.is_file(): + return {"ok": False, "error": f"File '{file_path}' not found in skill '{name}'"} + + try: + content = target.read_text(encoding="utf-8", errors="replace") + except OSError as exc: + return {"ok": False, "error": f"Failed to read file: {exc}"} + + max_chars = _skill_view_max_chars + truncated = len(content) > max_chars + content_out = content[:max_chars] + + result: Dict[str, Any] = { + "ok": True, + "name": name, + "file_path": file_path, + "content": content_out, + "path": str(target), + } + if truncated: + result["truncated"] = True + result["total_chars"] = len(content) + + return result diff --git a/src/leapflow/skills/index.py b/src/leapflow/skills/index.py index cd97bb6b..322281a9 100644 --- a/src/leapflow/skills/index.py +++ b/src/leapflow/skills/index.py @@ -16,6 +16,10 @@ logger = logging.getLogger(__name__) +# Package-bundled builtin SKILL.md skills — always scanned in addition to +# the user-profile skills_dir so they ship with the installed package. +_BUILTIN_SKILLS_DIR = Path(__file__).resolve().parent / "builtin_skills" + @dataclass(frozen=True) class SkillEntry: @@ -132,12 +136,41 @@ def _load_entries(self) -> List[SkillEntry]: return entries def _scan_skills_dir(self) -> List[SkillEntry]: - """L3: Full directory scan, parse all SKILL.md files.""" + """L3: Full directory scan, parse all SKILL.md files. + + Scans both the package-bundled builtin_skills directory and the + user-profile skills_dir. Builtin entries are scanned first; + user entries with the same name silently win (last-write-wins + in the name→entry mapping) so users can override builtins. + """ + seen_names: Dict[str, SkillEntry] = {} + + # 1) Package-bundled builtin skills + for entry in self._scan_single_dir(_BUILTIN_SKILLS_DIR): + seen_names[entry.name] = entry + + # 2) User-profile skills — may override builtins + for entry in self._scan_single_dir(self._skills_dir): + seen_names[entry.name] = entry + + # 3) MCP-bridged skills (auto-generated under _mcp_skills/) + mcp_skills_dir = self._skills_dir / "_mcp_skills" + for entry in self._scan_single_dir(mcp_skills_dir): + seen_names[entry.name] = entry + + entries = list(seen_names.values()) + logger.info( + "skill_index.scanned count=%d user_dir=%s builtin_dir=%s", + len(entries), self._skills_dir, _BUILTIN_SKILLS_DIR, + ) + return entries + + def _scan_single_dir(self, directory: Path) -> List[SkillEntry]: + """Scan one directory for SKILL.md subdirectories.""" entries: List[SkillEntry] = [] - if not self._skills_dir.exists(): + if not directory.exists(): return entries - - for skill_dir in sorted(self._skills_dir.iterdir()): + for skill_dir in sorted(directory.iterdir()): if not skill_dir.is_dir(): continue skill_md = skill_dir / "SKILL.md" @@ -146,10 +179,6 @@ def _scan_skills_dir(self) -> List[SkillEntry]: entry = self._parse_skill_md(skill_md, skill_dir) if entry is not None: entries.append(entry) - - logger.info( - "skill_index.scanned count=%d dir=%s", len(entries), self._skills_dir - ) return entries def _parse_skill_md(self, path: Path, skill_dir: Path) -> Optional[SkillEntry]: diff --git a/src/leapflow/skills/injector.py b/src/leapflow/skills/injector.py index 38fbb903..479fc86d 100644 --- a/src/leapflow/skills/injector.py +++ b/src/leapflow/skills/injector.py @@ -10,6 +10,8 @@ from pathlib import Path from typing import List, Optional +from leapflow.skills.index import _BUILTIN_SKILLS_DIR + logger = logging.getLogger(__name__) # Injection markers (LLM recognizes these as skill context) @@ -86,19 +88,31 @@ def build_injection_message( def find_skill_dir(self, name: str) -> Optional[Path]: """Resolve skill name -> directory path. + Searches user-profile skills_dir first (so user overrides win), + then the package-bundled builtin_skills directory. Attempts direct match first, then normalized (hyphen-based) lookup. """ - if not self._skills_dir.exists(): + # Search order: user dir first, then builtin + for base_dir in (self._skills_dir, _BUILTIN_SKILLS_DIR): + result = self._find_in_dir(base_dir, name) + if result is not None: + return result + return None + + @staticmethod + def _find_in_dir(base_dir: Path, name: str) -> Optional[Path]: + """Search a single directory for a skill by name.""" + if not base_dir.exists(): return None # Direct match - direct = self._skills_dir / name + direct = base_dir / name if direct.is_dir() and (direct / "SKILL.md").exists(): return direct # Normalized: replace spaces/underscores with hyphens normalized = name.lower().replace(" ", "-").replace("_", "-") - for d in self._skills_dir.iterdir(): + for d in base_dir.iterdir(): if not d.is_dir(): continue if d.name.lower().replace("_", "-") == normalized: diff --git a/src/leapflow/skills/mcp_bridge.py b/src/leapflow/skills/mcp_bridge.py new file mode 100644 index 00000000..fec4a1ac --- /dev/null +++ b/src/leapflow/skills/mcp_bridge.py @@ -0,0 +1,360 @@ +# Copyright (c) Alibaba, Inc. and its affiliates. +"""MCP-to-Skill bridge — present MCP server tool collections as discoverable Skills. + +Automatically wraps each connected MCP server's capability set into a generated +SKILL.md file so that the skill index, discovery tools, and LLM-facing skill +list can surface MCP-provided capabilities without manual authoring. + +Design: +- Reads tool metadata from the live ``McpServerManager`` (or any object that + satisfies the ``McpToolProvider`` Protocol defined here). +- Generates one SKILL.md per server under a ``_mcp_skills/`` subdirectory, + clearly separated from builtin and user skills. +- Registration is optional: if no MCP integration is available the bridge is a + silent no-op. +- All MCP imports are guarded so the module loads even when the ``mcp`` package + is absent. +""" +from __future__ import annotations + +import json +import logging +from dataclasses import dataclass, field +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Dict, List, Optional, Protocol, Sequence, runtime_checkable + +logger = logging.getLogger(__name__) + +# --------------------------------------------------------------------------- +# Auto-generation marker (placed at the top of every generated SKILL.md) +# --------------------------------------------------------------------------- +_AUTO_GEN_HEADER = ( + "<!-- AUTO-GENERATED by McpSkillBridge — do not edit manually. -->\n" + "<!-- Re-run bridge_all() to regenerate from live MCP server metadata. -->\n" +) + +_MCP_SKILLS_SUBDIR = "_mcp_skills" + + +# --------------------------------------------------------------------------- +# Protocol abstraction — decouples the bridge from any concrete MCP client +# --------------------------------------------------------------------------- +@runtime_checkable +class McpToolProvider(Protocol): + """Minimal contract for an object that can supply MCP tool metadata. + + Satisfied by ``McpManager`` in ``leapflow.platform.mcp_manager`` as well as + any test double. The bridge never calls tools — it only reads schemas. + """ + + def get_tool_schemas(self) -> List[Any]: ... + + +# --------------------------------------------------------------------------- +# Data types +# --------------------------------------------------------------------------- +@dataclass(frozen=True) +class McpSkillEntry: + """Compact record for a generated MCP skill.""" + + server_name: str + tool_count: int + generated_path: str + last_synced: str + + +@dataclass(frozen=True) +class _ToolInfo: + """Internal helper for rendering tool documentation.""" + + name: str + original_name: str + description: str + parameters: Dict[str, Any] = field(default_factory=dict) + read_only: bool = False + + +# --------------------------------------------------------------------------- +# McpSkillBridge +# --------------------------------------------------------------------------- +class McpSkillBridge: + """Bridge that generates SKILL.md files from connected MCP server tools. + + Usage:: + + bridge = McpSkillBridge(mcp_manager, skills_dir=layout.skills_dir) + entries = bridge.bridge_all() + bridge.register_with_index(skill_index) + + If *provider* is ``None`` or does not satisfy ``McpToolProvider``, every + public method degrades to a safe no-op. + """ + + def __init__( + self, + provider: Optional[McpToolProvider] = None, + *, + skills_dir: Optional[Path] = None, + ) -> None: + self._provider = provider if isinstance(provider, McpToolProvider) else None + self._skills_dir = Path(skills_dir) if skills_dir else None + self._entries: List[McpSkillEntry] = [] + + # ------------------------------------------------------------------ + # Public API + # ------------------------------------------------------------------ + + def discover_servers(self) -> Dict[str, List[_ToolInfo]]: + """Return a mapping of *server_name* -> list of tool info. + + Groups the flat tool list from the provider by ``server_name``. + Returns an empty dict when the provider is unavailable. + """ + if self._provider is None: + return {} + + try: + schemas = self._provider.get_tool_schemas() + except Exception as exc: + logger.warning("mcp_bridge: failed to query tool schemas: %s", exc) + return {} + + servers: Dict[str, List[_ToolInfo]] = {} + for schema in schemas: + info = self._schema_to_info(schema) + if info is None: + continue + sname = getattr(schema, "server_name", "unknown") + servers.setdefault(sname, []).append(info) + return servers + + def generate_skill_md(self, server_name: str) -> Optional[str]: + """Generate SKILL.md content for a single MCP server. + + Returns the Markdown string, or ``None`` if the server has no tools. + """ + servers = self.discover_servers() + tools = servers.get(server_name) + if not tools: + return None + return self._render_skill_md(server_name, tools) + + def bridge_all(self) -> List[McpSkillEntry]: + """Generate SKILL.md files for every connected MCP server. + + Writes files under ``<skills_dir>/_mcp_skills/<server_name>/SKILL.md`` + and returns a list of ``McpSkillEntry`` records. + """ + if self._provider is None or self._skills_dir is None: + return [] + + servers = self.discover_servers() + if not servers: + return [] + + mcp_dir = self._skills_dir / _MCP_SKILLS_SUBDIR + mcp_dir.mkdir(parents=True, exist_ok=True) + + entries: List[McpSkillEntry] = [] + now_iso = datetime.now(timezone.utc).isoformat(timespec="seconds") + + for server_name, tools in servers.items(): + content = self._render_skill_md(server_name, tools) + if content is None: + continue + skill_dir = mcp_dir / _sanitize_dir_name(server_name) + skill_dir.mkdir(parents=True, exist_ok=True) + skill_md = skill_dir / "SKILL.md" + skill_md.write_text(content, encoding="utf-8") + logger.info( + "mcp_bridge: generated skill for server '%s' (%d tools) at %s", + server_name, len(tools), skill_md, + ) + entries.append( + McpSkillEntry( + server_name=server_name, + tool_count=len(tools), + generated_path=str(skill_md), + last_synced=now_iso, + ) + ) + + self._entries = entries + return entries + + def register_with_index(self, skill_index: Any) -> int: + """Invalidate the skill index so generated MCP skills are picked up. + + The ``SkillIndex`` scans ``_mcp_skills/`` as sub-directories of + ``skills_dir``, so we just need to bust the cache. Returns the + number of entries that were bridged. + """ + if hasattr(skill_index, "invalidate"): + skill_index.invalidate() + return len(self._entries) + + @property + def entries(self) -> List[McpSkillEntry]: + """Previously bridged entries (populated after ``bridge_all()``).""" + return list(self._entries) + + # ------------------------------------------------------------------ + # Rendering + # ------------------------------------------------------------------ + + def _render_skill_md( + self, server_name: str, tools: Sequence[_ToolInfo] + ) -> Optional[str]: + """Build the full SKILL.md text for one MCP server.""" + if not tools: + return None + + tool_names = [t.name for t in tools] + tags = _derive_tags(server_name, tools) + triggers = _derive_triggers(server_name, tools) + + # ---- YAML frontmatter ---- + frontmatter_lines = [ + "---", + f"name: mcp_{_sanitize_dir_name(server_name)}", + f'description: "Auto-generated skill for MCP server: {server_name}"', + "version: 0.0.0", + "metadata:", + " leapflow:", + ' category: "integration"', + ' source: "mcp"', + " confidence: 0.8", + " quality_score: 0.9", + " hermes:", + ' category: "integration"', + f" tags: {json.dumps(tags)}", + f" requires_tools: {json.dumps(tool_names)}", + "platforms: []", + "triggers:", + ] + for trig in triggers: + frontmatter_lines.append(f' - "{trig}"') + frontmatter_lines.append("---") + frontmatter = "\n".join(frontmatter_lines) + + # ---- Body ---- + body_parts: List[str] = [ + _AUTO_GEN_HEADER, + f"# {server_name} (MCP Server)\n", + "## Overview\n", + ( + f"This skill provides access to the **{server_name}** MCP server, " + f"which exposes {len(tools)} tool(s). The tools listed below are " + "available for use once the MCP server connection is established.\n" + ), + "> **Note:** This skill requires an active MCP server connection. " + "Ensure the server is configured and running before invoking these tools.\n", + "## Available Tools\n", + ] + + for tool in tools: + body_parts.append(f"### `{tool.name}`\n") + if tool.description: + body_parts.append(f"{tool.description}\n") + if tool.original_name != tool.name: + body_parts.append(f"*Original name:* `{tool.original_name}`\n") + ro_label = "read-only" if tool.read_only else "may have side effects" + body_parts.append(f"*Effect:* {ro_label}\n") + if tool.parameters and tool.parameters.get("properties"): + body_parts.append("**Parameters:**\n") + body_parts.append("| Name | Type | Required | Description |") + body_parts.append("|------|------|----------|-------------|") + props = tool.parameters.get("properties", {}) + required_set = set(tool.parameters.get("required", [])) + for pname, pschema in props.items(): + ptype = pschema.get("type", "any") + pdesc = pschema.get("description", "") + req = "Yes" if pname in required_set else "No" + body_parts.append(f"| `{pname}` | {ptype} | {req} | {pdesc} |") + body_parts.append("") + + # ---- Workflow template ---- + body_parts.append("## Workflow\n") + body_parts.append( + f"To accomplish tasks with the **{server_name}** server:\n" + ) + body_parts.append( + "1. Verify the MCP server connection is active.\n" + "2. Identify the appropriate tool from the list above.\n" + "3. Construct the required parameters based on the schema.\n" + "4. Call the tool and interpret the result.\n" + "5. For multi-step operations, chain tool calls in logical order.\n" + ) + + body_parts.append("## Error Handling\n") + body_parts.append( + "| Situation | Action |\n" + "|---|---|\n" + "| MCP server not connected | Report connection failure; suggest checking server config. |\n" + "| Tool call times out | Retry once, then report timeout to user. |\n" + "| Unknown tool name | List available tools and suggest closest match. |\n" + ) + + body = "\n".join(body_parts) + return f"{frontmatter}\n\n{body}" + + # ------------------------------------------------------------------ + # Helpers + # ------------------------------------------------------------------ + + @staticmethod + def _schema_to_info(schema: Any) -> Optional[_ToolInfo]: + """Convert a provider schema object to ``_ToolInfo``.""" + try: + return _ToolInfo( + name=getattr(schema, "name", ""), + original_name=getattr(schema, "original_name", getattr(schema, "name", "")), + description=getattr(schema, "description", ""), + parameters=getattr(schema, "parameters", {}), + read_only=getattr(schema, "read_only", False), + ) + except Exception: + return None + + +# --------------------------------------------------------------------------- +# Module-level helpers +# --------------------------------------------------------------------------- + +def _sanitize_dir_name(name: str) -> str: + """Produce a filesystem-safe directory name from a server name.""" + return name.replace(" ", "_").replace("-", "_").replace(".", "_").lower() + + +def _derive_tags(server_name: str, tools: Sequence[_ToolInfo]) -> List[str]: + """Derive a compact tag list from server name and tool descriptions.""" + tags: List[str] = ["mcp", server_name.lower().replace(" ", "-")] + # Pick up to 3 unique keywords from tool descriptions + seen: set[str] = set(tags) + for tool in tools[:10]: + for word in tool.description.split()[:5]: + clean = word.strip(".,;:()").lower() + if len(clean) > 3 and clean not in seen: + tags.append(clean) + seen.add(clean) + if len(tags) >= 6: + break + if len(tags) >= 6: + break + return tags + + +def _derive_triggers( + server_name: str, tools: Sequence[_ToolInfo] +) -> List[str]: + """Derive trigger phrases for the generated skill.""" + triggers = [ + f"use {server_name}", + f"{server_name} tools", + ] + # Add up to 3 triggers from tool names + for tool in tools[:3]: + if tool.original_name: + triggers.append(tool.original_name.replace("_", " ")) + return triggers diff --git a/src/leapflow/skills/tool_call_parser.py b/src/leapflow/skills/tool_call_parser.py new file mode 100644 index 00000000..a752fe95 --- /dev/null +++ b/src/leapflow/skills/tool_call_parser.py @@ -0,0 +1,137 @@ +# Copyright (c) Alibaba, Inc. and its affiliates. +"""Tool call parsing utilities for LLM response content. + +Extracts structured tool calls from free-form LLM text output. Supports +multiple formats: markdown JSON code blocks, XML-style ``<tool_call>`` +wrappers, inline JSON with ``"name"``/``"tool"`` keys, and legacy +``{"tool": ..., "params": {...}}`` format. + +Also provides a repetition detector that aborts when the LLM is stuck +in a degenerate output loop. + +These utilities are shared between the legacy ReAct skill executor +(``tool_executor.py``) and the unified agent tool dispatch engine +(``tool_dispatch_engine.py``). +""" + +from __future__ import annotations + +import json +import re +from typing import Optional + +from leapflow.skills.tool_types import ToolCall + +_TOOL_CALL_PATTERN = re.compile( + r"```(?:json)?\s*(\{.*?\})\s*```", re.DOTALL +) +_INLINE_JSON_PATTERN = re.compile( + r'\{\s*"name"\s*:', re.DOTALL +) + + +def detect_repetition(content: str, threshold: int = 10) -> bool: + """Detect if LLM output is stuck in a repetitive pattern.""" + if len(content) < 100: + return False + # Check for repeated closing tags (common failure mode) + repeated_patterns = ["</invoke>", "</tool_call>", "```\n```"] + for pattern in repeated_patterns: + if content.count(pattern) >= threshold: + return True + # Check last 200 chars for character-level repetition + tail = content[-200:] + if len(set(tail.split())) <= 3 and len(tail) > 50: + return True + return False + + +def parse_tool_call(content: str) -> Optional[ToolCall]: + """Extract a tool call JSON from LLM response text. + + Supports: + - ```json {"name": ..., "arguments": {...}} ``` (primary) + - Inline {"name": ...} patterns + - <tool_call>{"name": ..., "arguments": {...}}</tool_call> patterns + - Legacy {"tool": ..., "params": {...}} format + """ + # 1. Standard markdown code block + match = _TOOL_CALL_PATTERN.search(content) + if match: + result = _try_parse_json(match.group(1)) + if result: + return result + + # 2. <tool_call> XML-style wrapper + tc_match = re.search( + r'<tool_call>\s*(\{.*?\})\s*(?:</tool_call>|</invoke>)', content, re.DOTALL + ) + if tc_match: + result = _try_parse_json(tc_match.group(1)) + if result: + return result + + # 3. Inline JSON with "name" or "tool" key + idx = -1 + for pattern_str in ['"name"', '"tool"']: + search = content.find('{') + while search != -1: + # Check if this { starts a valid tool call JSON + if pattern_str in content[search:search + 50]: + idx = search + break + search = content.find('{', search + 1) + if idx != -1: + break + + if idx == -1: + match2 = _INLINE_JSON_PATTERN.search(content) + if match2: + idx = match2.start() + + if idx != -1: + depth = 0 + end = idx + for i in range(idx, min(len(content), idx + 2000)): # limit scan to 2000 chars + if content[i] == '{': + depth += 1 + elif content[i] == '}': + depth -= 1 + if depth == 0: + end = i + 1 + break + if depth == 0: + return _try_parse_json(content[idx:end]) + + return None + + +def _try_parse_json(text: str) -> Optional[ToolCall]: + """Parse a JSON string into a ToolCall (OpenAI function calling format).""" + try: + data = json.loads(text) + if not isinstance(data, dict): + return None + + # Primary: OpenAI function calling format {"name": ..., "arguments": {...}} + if "name" in data: + name = data["name"] + params = data.get("arguments", data.get("params", data.get("parameters", {}))) + if isinstance(params, str): + # Sometimes arguments is a JSON string + try: + params = json.loads(params) + except (json.JSONDecodeError, TypeError): + params = {"raw": params} + return ToolCall(name=str(name), params=params if isinstance(params, dict) else {}) + + # Fallback: legacy {"tool": ..., "params": {...}} format + if "tool" in data: + return ToolCall( + name=data["tool"], + params=data.get("params", data.get("arguments", data.get("parameters", {}))) or {}, + ) + + except (json.JSONDecodeError, KeyError, TypeError): + pass + return None diff --git a/src/leapflow/skills/tool_executor.py b/src/leapflow/skills/tool_executor.py index 4a7724bd..611a63dc 100644 --- a/src/leapflow/skills/tool_executor.py +++ b/src/leapflow/skills/tool_executor.py @@ -1,16 +1,24 @@ # Copyright (c) Alibaba, Inc. and its affiliates. -"""ReAct-style tool-use executor for SKILL.md skills. +"""**DEPRECATED** — Legacy ReAct-style tool-use executor for SKILL.md skills. + +.. deprecated:: 0.3.0 + ``ToolUseSkillExecutor`` and ``ExecutionToolset`` are legacy components + that predate the Hermes-inspired unified agent loop. New code should use + ``SkillActivator`` / ``SkillInjector`` / ``SkillDispatcher`` and the + unified tool dispatch engine (``engine.tool_dispatch_engine``) instead. + + Domain types (``ToolCall``, ``ToolDefinition``, ``StepOutput``, + ``ExecutionPort``) have been relocated to ``leapflow.skills.tool_types``. + Parsing utilities (``parse_tool_call``, ``detect_repetition``) have been + relocated to ``leapflow.skills.tool_call_parser``. + + Imports from this module are forwarded for backward compatibility but will + be removed in a future release. Gives the LLM access to real system tools (file ops, shell, UI) via ExecutionPort. Each SKILL.md instruction is executed as a bounded observe → reason → act loop. -Architecture: - ToolDefinition → describes available tools for LLM prompt - ToolCall → parsed from LLM JSON output - ExecutionToolset → dispatches ToolCalls to ExecutionPort (SRP: only routing) - ToolUseSkillExecutor → orchestrates the ReAct loop per instruction - ``ExecutionToolset`` is the desktop execution tool container used exclusively by the bounded ReAct skill executor (SKILL.md instructions and the chat desktop-action fallback). It is NOT the unified agent dispatch surface — @@ -24,13 +32,28 @@ import logging import re import shlex +import warnings from dataclasses import dataclass, field -from typing import TYPE_CHECKING, Any, Dict, List, Optional, Protocol, runtime_checkable +from typing import TYPE_CHECKING, Any, Dict, List, Optional from leapflow.engine.budget import BudgetConfig, BudgetStatus, IterationBudget from leapflow.engine.context.context_compressor import CompressorConfig, ContextCompressor from leapflow.engine.message_healer import MessageHealer +# Re-export domain types from their canonical home for backward compatibility. +from leapflow.skills.tool_types import ( # noqa: F401 — re-export + ExecutionPort, + StepOutput, + ToolCall, + ToolDefinition, +) + +# Re-export parsing utilities from their canonical home. +from leapflow.skills.tool_call_parser import ( # noqa: F401 — re-export + detect_repetition as _detect_repetition_impl, + parse_tool_call as _parse_tool_call_impl, +) + if TYPE_CHECKING: from leapflow.engine.confirmation import IOProvider from leapflow.llm.base import LLMProvider @@ -39,6 +62,21 @@ logger = logging.getLogger(__name__) +# Module-level deprecation notice: emitted once when the module is first imported. +warnings.warn( + "leapflow.skills.tool_executor is deprecated. " + "Use leapflow.skills.tool_types for domain types, " + "leapflow.skills.tool_call_parser for parsing utilities, " + "and the unified engine loop for execution.", + DeprecationWarning, + stacklevel=2, +) + +# Backward-compat aliases for private parser functions previously in this module. +_detect_repetition = _detect_repetition_impl +_parse_tool_call = _parse_tool_call_impl + + _DRIFT_THRESHOLD = 2 _MUTATING_SHELL_PREFIXES = ( "open", "rm", "mv", "cp", "mkdir", "touch", "chmod", "chown", @@ -50,12 +88,6 @@ "sed", "awk", "tee", ) _SHELL_CHAIN_SPLIT = re.compile(r"\s*(?:&&|\|\||;)\s*") -_TOOL_CALL_PATTERN = re.compile( - r"```(?:json)?\s*(\{.*?\})\s*```", re.DOTALL -) -_INLINE_JSON_PATTERN = re.compile( - r'\{\s*"name"\s*:', re.DOTALL -) def _is_shell_mutating(command: str) -> bool: @@ -72,65 +104,14 @@ def _is_shell_mutating(command: str) -> bool: return False -@dataclass(frozen=True) -class ToolDefinition: - """Schema for one available tool — injected into the LLM system prompt. - - Traits: - mutates_state: Tool changes observable state → clears dedup cache. - counts_as_progress: Tool represents forward progress toward the goal - → triggers completion HINT. Defaults to mutates_state. - Set False for timing/polling tools (wait, wait_until_stable). - """ - - name: str - description: str - parameters: Dict[str, str] - mutates_state: bool = False - counts_as_progress: bool | None = None - - @property - def is_progress(self) -> bool: - if self.counts_as_progress is not None: - return self.counts_as_progress - return self.mutates_state - - -@dataclass(frozen=True) -class ToolCall: - """Parsed tool invocation from LLM output.""" - - name: str - params: Dict[str, Any] - - -@dataclass -class StepOutput: - """Result of executing one instruction step.""" - - ok: bool - result: str = "" - error: str = "" - tool_calls_made: int = 0 - goal_complete: bool = False - - -@runtime_checkable -class ExecutionPort(Protocol): - """Minimal execution interface (matches vsi.ports.ExecutionPort).""" - - async def perform_file_op(self, op: str, params: Dict[str, Any]) -> Dict[str, Any]: ... - async def exec_shell(self, command: str) -> Dict[str, Any]: ... - async def launch_app( - self, app_id: str, urls: Optional[List[str]] = None - ) -> Dict[str, Any]: ... - async def perform_ui_action( - self, node_id: str, action: str, params: Optional[Dict[str, Any]] = None - ) -> Dict[str, Any]: ... +# Domain types are now defined in leapflow.skills.tool_types and re-exported +# at the top of this module. The class definitions below have been removed; +# ToolDefinition, ToolCall, StepOutput, ExecutionPort are available via the +# re-export imports above. # ═══════════════════════════════════════════════════════════════════════ -# ExecutionToolset — desktop execution dispatch layer +# ExecutionToolset — desktop execution dispatch layer (DEPRECATED) # ═══════════════════════════════════════════════════════════════════════ @@ -155,6 +136,11 @@ def __init__( class ExecutionToolset: """Maps desktop tool names to ExecutionPort methods via a handler registry. + .. deprecated:: 0.3.0 + This class is part of the legacy ReAct skill executor. New code should + use the unified tool dispatch engine (``engine.tool_dispatch_engine``) + and the ``ToolPluginRegistry`` handler system instead. + Serves the bounded ReAct skill executor: registers ExecutionPort-derived defaults (file ops, shell, launch_app, ui_action, done) plus optional semantic UI tools (via ``build_execution_toolset``). The unified agent @@ -170,6 +156,12 @@ def __init__( policy: Optional["PolicyEngine"] = None, io: Optional["IOProvider"] = None, ) -> None: + warnings.warn( + "ExecutionToolset is deprecated. Use the unified tool dispatch " + "engine and ToolPluginRegistry handlers instead.", + DeprecationWarning, + stacklevel=2, + ) self._execution = execution self.policy = policy self.io = io @@ -496,7 +488,14 @@ def _demote_stderr(result: Dict[str, Any]) -> Dict[str, Any]: class ToolUseSkillExecutor: - """Executes SKILL.md instructions via bounded ReAct loop with real tools.""" + """Executes SKILL.md instructions via bounded ReAct loop with real tools. + + .. deprecated:: 0.3.0 + This executor is a legacy component that predates the Hermes-inspired + unified agent loop. New code should use ``SkillActivator`` / + ``SkillInjector`` / ``SkillDispatcher`` and the unified tool dispatch + engine instead. + """ def __init__( self, @@ -512,6 +511,12 @@ def __init__( compressor_config: Optional[CompressorConfig] = None, step_timeout_s: float = 30.0, ) -> None: + warnings.warn( + "ToolUseSkillExecutor is deprecated. Use the unified agent loop " + "(SkillActivator / SkillDispatcher) instead.", + DeprecationWarning, + stacklevel=2, + ) self._llm = llm self._vlm = vlm self._toolset = toolset @@ -964,111 +969,6 @@ def _format_results(self, results: List[StepOutput]) -> str: return "\n".join(parts) -# ═══════════════════════════════════════════════════════════════════════ -# Tool call parsing -# ═══════════════════════════════════════════════════════════════════════ - - -def _detect_repetition(content: str, threshold: int = 10) -> bool: - """Detect if LLM output is stuck in a repetitive pattern.""" - if len(content) < 100: - return False - # Check for repeated closing tags (common failure mode) - repeated_patterns = ["</invoke>", "</tool_call>", "```\n```"] - for pattern in repeated_patterns: - if content.count(pattern) >= threshold: - return True - # Check last 200 chars for character-level repetition - tail = content[-200:] - if len(set(tail.split())) <= 3 and len(tail) > 50: - return True - return False - - -def _parse_tool_call(content: str) -> Optional[ToolCall]: - """Extract a tool call JSON from LLM response text. - - Supports: - - ```json {"name": ..., "arguments": {...}} ``` (primary) - - Inline {"name": ...} patterns - - <tool_call>{"name": ..., "arguments": {...}}</tool_call> patterns - - Legacy {"tool": ..., "params": {...}} format - """ - # 1. Standard markdown code block - match = _TOOL_CALL_PATTERN.search(content) - if match: - result = _try_parse_json(match.group(1)) - if result: - return result - - # 2. <tool_call> XML-style wrapper - tc_match = re.search(r'<tool_call>\s*(\{.*?\})\s*(?:</tool_call>|</invoke>)', content, re.DOTALL) - if tc_match: - result = _try_parse_json(tc_match.group(1)) - if result: - return result - - # 3. Inline JSON with "name" or "tool" key - idx = -1 - for pattern_str in ['"name"', '"tool"']: - search = content.find('{') - while search != -1: - # Check if this { starts a valid tool call JSON - if pattern_str in content[search:search + 50]: - idx = search - break - search = content.find('{', search + 1) - if idx != -1: - break - - if idx == -1: - match2 = _INLINE_JSON_PATTERN.search(content) - if match2: - idx = match2.start() - - if idx != -1: - depth = 0 - end = idx - for i in range(idx, min(len(content), idx + 2000)): # limit scan to 2000 chars - if content[i] == '{': - depth += 1 - elif content[i] == '}': - depth -= 1 - if depth == 0: - end = i + 1 - break - if depth == 0: - return _try_parse_json(content[idx:end]) - - return None - - -def _try_parse_json(text: str) -> Optional[ToolCall]: - """Parse a JSON string into a ToolCall (OpenAI function calling format).""" - try: - data = json.loads(text) - if not isinstance(data, dict): - return None - - # Primary: OpenAI function calling format {"name": ..., "arguments": {...}} - if "name" in data: - name = data["name"] - params = data.get("arguments", data.get("params", data.get("parameters", {}))) - if isinstance(params, str): - # Sometimes arguments is a JSON string - try: - params = json.loads(params) - except (json.JSONDecodeError, TypeError): - params = {"raw": params} - return ToolCall(name=str(name), params=params if isinstance(params, dict) else {}) - - # Fallback: legacy {"tool": ..., "params": {...}} format - if "tool" in data: - return ToolCall( - name=data["tool"], - params=data.get("params", data.get("arguments", data.get("parameters", {}))), - ) - - except (json.JSONDecodeError, KeyError, TypeError): - pass - return None +# Parser functions have been relocated to leapflow.skills.tool_call_parser. +# The module-level backward-compat aliases (_detect_repetition, _parse_tool_call) +# are defined near the top of this file via the re-export imports. diff --git a/src/leapflow/skills/tool_types.py b/src/leapflow/skills/tool_types.py new file mode 100644 index 00000000..cff7c13b --- /dev/null +++ b/src/leapflow/skills/tool_types.py @@ -0,0 +1,72 @@ +# Copyright (c) Alibaba, Inc. and its affiliates. +"""Shared domain types for skill tool execution. + +Contains the lightweight data classes and protocols used across the skill +subsystem — tool definitions, parsed tool calls, step outputs, and the +execution port contract. These types are intentionally decoupled from any +executor implementation so that modules like ``action_policy``, +``semantic_schema``, and ``tool_dispatch_engine`` can import them without +pulling in the legacy ReAct executor. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any, Dict, List, Optional, Protocol, runtime_checkable + + +@dataclass(frozen=True) +class ToolDefinition: + """Schema for one available tool — injected into the LLM system prompt. + + Traits: + mutates_state: Tool changes observable state -> clears dedup cache. + counts_as_progress: Tool represents forward progress toward the goal + -> triggers completion HINT. Defaults to mutates_state. + Set False for timing/polling tools (wait, wait_until_stable). + """ + + name: str + description: str + parameters: Dict[str, str] + mutates_state: bool = False + counts_as_progress: bool | None = None + + @property + def is_progress(self) -> bool: + if self.counts_as_progress is not None: + return self.counts_as_progress + return self.mutates_state + + +@dataclass(frozen=True) +class ToolCall: + """Parsed tool invocation from LLM output.""" + + name: str + params: Dict[str, Any] + + +@dataclass +class StepOutput: + """Result of executing one instruction step.""" + + ok: bool + result: str = "" + error: str = "" + tool_calls_made: int = 0 + goal_complete: bool = False + + +@runtime_checkable +class ExecutionPort(Protocol): + """Minimal execution interface (matches vsi.ports.ExecutionPort).""" + + async def perform_file_op(self, op: str, params: Dict[str, Any]) -> Dict[str, Any]: ... + async def exec_shell(self, command: str) -> Dict[str, Any]: ... + async def launch_app( + self, app_id: str, urls: Optional[List[str]] = None + ) -> Dict[str, Any]: ... + async def perform_ui_action( + self, node_id: str, action: str, params: Optional[Dict[str, Any]] = None + ) -> Dict[str, Any]: ... diff --git a/src/leapflow/tools/hub_tool.py b/src/leapflow/tools/hub_tool.py index 50c334f1..cd3e7eb3 100644 --- a/src/leapflow/tools/hub_tool.py +++ b/src/leapflow/tools/hub_tool.py @@ -179,12 +179,14 @@ async def hub_pull_tool( async def hub_search_tool( query: str = "", + federated: bool = False, **kwargs: Any, ) -> str: """Search for skills on the Hub. Returns formatted results. Args: query: Free-text search query. + federated: If True, search across all registered backends in parallel. """ from leapflow.config import get_settings from leapflow.hub import HubClient @@ -198,20 +200,35 @@ async def hub_search_tool( default_owner=settings.hub_default_owner, default_visibility=settings.hub_default_visibility, repo_prefix=settings.hub_repo_prefix, + search_sources=getattr(settings, "hub_search_sources", ""), ) try: - results = await client.search(query) + if federated: + fed_result = await client.federated_search(query) + results = list(fed_result.results) + extra = "" + if fed_result.failed_backends: + extra = ( + f"\n(backends failed: {', '.join(fed_result.failed_backends)})" + ) + else: + results = await client.search(query) + extra = "" except Exception as e: return f"Search failed: {type(e).__name__}: {e}" if not results: return f"No skills found for '{query}'." - lines = [f"Found {len(results)} skill(s) for '{query}':\n"] + mode_label = "federated " if federated else "" + lines = [f"Found {len(results)} skill(s) via {mode_label}search for '{query}':"] for r in results: - desc = f" — {r.description}" if r.description else "" - lines.append(f" {r.repo_id} v{r.version}{desc}") + desc = f" \u2014 {r.description}" if r.description else "" + hub_tag = f" [{r.hub_type}]" if r.hub_type else "" + lines.append(f" {r.repo_id} v{r.version}{hub_tag}{desc}") + if extra: + lines.append(extra) return "\n".join(lines) @@ -284,6 +301,195 @@ async def hub_sync_tool( return "\n".join(lines) +async def hub_federated_search_tool( + query: str = "", + **kwargs: Any, +) -> str: + """Search for skills across ALL registered Hub backends in parallel. + + Queries every configured backend concurrently, deduplicates by skill name + (preferring higher version / more downloads), and returns a unified listing. + + Args: + query: Free-text search query. + """ + return await hub_search_tool(query=query, federated=True, **kwargs) + + +# ─── Marketplace & Contribution Tool Implementations ───────────────────────── + + +async def hub_marketplace_browse_tool( + category: str = "", + featured: bool = False, + trending: bool = False, + limit: int = 10, + **kwargs: Any, +) -> str: + """Browse marketplace categories and featured/trending skills. + + Args: + category: Filter by category name (empty = show categories overview). + featured: If True, show only featured/trending entries. + trending: If True, show top skills by installs/downloads. + limit: Maximum number of results (default 10). + """ + marketplace = _get_marketplace(**kwargs) + if marketplace is None: + return "Error: Hub client not available." + + if not category and not featured and not trending: + # Show categories overview + cats = marketplace.get_categories() + lines = ["Skill Marketplace Categories:"] + for c in cats: + count = f" ({c.skill_count} skills)" if c.skill_count else "" + lines.append(f" {c.icon} {c.name}{count} — {c.description}") + lines.append("\nUse category filter to browse skills in a specific category.") + return "\n".join(lines) + + if featured: + entries = marketplace.get_featured() + label = "Featured & Trending" + elif trending: + entries = marketplace.get_trending(limit=limit) + label = f"Top {limit} Trending" + elif category: + entries = marketplace.get_by_category(category) + label = f"Category: {category}" + else: + entries = [] + label = "Browse" + + if not entries: + return f"No skills found for {label}." + + lines = [f"{label} ({len(entries)} skill{'s' if len(entries) != 1 else ''}):\n"] + for e in entries[:limit]: + flags = [] + if e.featured: + flags.append("★ featured") + if e.trending: + flags.append("🔥 trending") + flag_str = f" [{', '.join(flags)}]" if flags else "" + desc = f" — {e.description}" if e.description else "" + lines.append(f" {e.repo_id} v{e.version}{flag_str}{desc}") + return "\n".join(lines) + + +async def hub_marketplace_install_tool( + name: str = "", + **kwargs: Any, +) -> str: + """Install a skill from the marketplace. + + Args: + name: Skill name or repo_id to install. + """ + if not name: + return "Error: name is required." + + marketplace = _get_marketplace(**kwargs) + if marketplace is None: + return "Error: Hub client not available." + + return await marketplace.install(name) + + +async def hub_contribute_tool( + skill_name: str = "", + action: str = "submit", + hub_type: str = "github", + **kwargs: Any, +) -> str: + """Submit a skill to the community or check contribution status. + + Args: + skill_name: Name of the local skill. + action: 'prepare', 'submit', 'status', or 'list'. + hub_type: Target hub backend for submission (default: 'github'). + """ + contributor = _get_contributor(**kwargs) + if contributor is None: + return "Error: Hub client not available." + + ctx = kwargs.get("ctx") + + if action == "list": + records = contributor.list_my_contributions() + if not records: + return "No contributions found." + lines = [f"Your Contributions ({len(records)}):"] + for r in records: + lines.append(f" {r.skill_name} — {r.status} ({r.hub_type or 'local'})") + return "\n".join(lines) + + if not skill_name: + return "Error: skill_name is required." + + if action == "prepare": + return await contributor.prepare(skill_name, ctx=ctx) + elif action == "submit": + return await contributor.submit(skill_name, hub_type=hub_type, ctx=ctx) + elif action == "status": + try: + status = contributor.check_status(skill_name) + return f"Contribution status for '{skill_name}': {status.value}" + except KeyError as exc: + return str(exc) + else: + return f"Unknown action '{action}'. Use 'prepare', 'submit', 'status', or 'list'." + + +def _get_marketplace(**kwargs: Any) -> Any: + """Build a SkillMarketplace from runtime context.""" + from leapflow.config import get_settings + from leapflow.hub import HubClient + from leapflow.hub.marketplace import SkillMarketplace + + try: + settings = get_settings() + client = HubClient( + hub_type=settings.hub_type, + default_owner=settings.hub_default_owner, + default_visibility=settings.hub_default_visibility, + repo_prefix=settings.hub_repo_prefix, + search_sources=getattr(settings, "hub_search_sources", ""), + ) + profile_layout = getattr(settings, "profile_layout", None) + cache_path = None + if profile_layout is not None: + cache_path = profile_layout.root / "marketplace_cache.json" + return SkillMarketplace(client, cache_path=cache_path) + except Exception as exc: + logger.warning("Failed to create marketplace: %s", exc) + return None + + +def _get_contributor(**kwargs: Any) -> Any: + """Build a CommunityContributor from runtime context.""" + from leapflow.config import get_settings + from leapflow.hub import HubClient + from leapflow.hub.contribute import CommunityContributor + + try: + settings = get_settings() + client = HubClient( + hub_type=settings.hub_type, + default_owner=settings.hub_default_owner, + default_visibility=settings.hub_default_visibility, + repo_prefix=settings.hub_repo_prefix, + ) + profile_layout = getattr(settings, "profile_layout", None) + store_path = None + if profile_layout is not None: + store_path = profile_layout.root / "contributions.json" + return CommunityContributor(client, store_path=store_path) + except Exception as exc: + logger.warning("Failed to create contributor: %s", exc) + return None + + # ─── Tool Definitions (OpenAI function calling schema) ─────────────────────── @@ -352,6 +558,40 @@ async def hub_sync_tool( "function": { "name": "hub_search", "description": "Search for skills on the Hub by keyword or description.", + "parameters": { + "type": "object", + "properties": { + "query": { + "type": "string", + "description": "Free-text search query for finding skills", + }, + "federated": { + "type": "boolean", + "description": ( + "If true, search across all registered backends " + "in parallel (default: false)" + ), + }, + }, + "required": ["query"], + }, + "x_leapflow": { + "category": "hub", + "risk_level": "read_only", + "schema_cost": "high", + "requires_approval": False, + }, + }, + }, + { + "type": "function", + "function": { + "name": "hub_federated_search", + "description": ( + "Search for skills across ALL registered Hub backends in parallel. " + "Queries every configured backend concurrently, deduplicates results, " + "and returns a unified listing sorted by relevance." + ), "parameters": { "type": "object", "properties": { @@ -397,6 +637,106 @@ async def hub_sync_tool( }, }, }, + { + "type": "function", + "function": { + "name": "hub_marketplace_browse", + "description": ( + "Browse the skill marketplace — view categories, featured skills, " + "and trending skills. Supports category filtering." + ), + "parameters": { + "type": "object", + "properties": { + "category": { + "type": "string", + "description": ( + "Filter by category name " + "(development, research, automation, productivity, " + "security, operations, analysis, integration). " + "Empty shows overview." + ), + }, + "featured": { + "type": "boolean", + "description": "Show only featured/trending entries (default: false)", + }, + "trending": { + "type": "boolean", + "description": "Show top skills by installs/downloads (default: false)", + }, + "limit": { + "type": "integer", + "description": "Maximum results to return (default: 10)", + }, + }, + }, + "x_leapflow": { + "category": "hub", + "risk_level": "read_only", + "schema_cost": "high", + "requires_approval": False, + }, + }, + }, + { + "type": "function", + "function": { + "name": "hub_marketplace_install", + "description": "Install a skill from the marketplace by name or repo_id.", + "parameters": { + "type": "object", + "properties": { + "name": { + "type": "string", + "description": "Skill name or repo_id to install", + }, + }, + "required": ["name"], + }, + "x_leapflow": { + "category": "hub", + "risk_level": "medium", + "schema_cost": "high", + "requires_approval": True, + }, + }, + }, + { + "type": "function", + "function": { + "name": "hub_contribute", + "description": ( + "Submit a local skill to the community hub, check contribution " + "status, or list all your contributions." + ), + "parameters": { + "type": "object", + "properties": { + "skill_name": { + "type": "string", + "description": "Name of the local skill to contribute", + }, + "action": { + "type": "string", + "enum": ["prepare", "submit", "status", "list"], + "description": "Contribution action (default: submit)", + }, + "hub_type": { + "type": "string", + "description": "Target hub backend (default: github)", + }, + }, + }, + "x_leapflow": { + "category": "hub", + "risk_level": "medium", + "schema_cost": "high", + "requires_approval": True, + "mutates_state": True, + }, + }, + }, ] @@ -407,9 +747,9 @@ async def hub_sync_tool( "name": "hub_push", "description": "Push a local skill to the Hub for sharing or backup.", "parameters": { - "skill_name": "string (required) — name of the skill to push", - "visibility": "string (optional) — 'private' (default), 'public', or 'internal'", - "version": "string (optional) — version override", + "skill_name": "string (required) \u2014 name of the skill to push", + "visibility": "string (optional) \u2014 'private' (default), 'public', or 'internal'", + "version": "string (optional) \u2014 version override", }, "handler": hub_push_tool, "mutates_state": True, @@ -418,8 +758,8 @@ async def hub_sync_tool( "name": "hub_pull", "description": "Pull a skill from the Hub to install locally.", "parameters": { - "repo_id": "string (required) — repository identifier", - "version": "string (optional) — specific version to pull", + "repo_id": "string (required) \u2014 repository identifier", + "version": "string (optional) \u2014 specific version to pull", }, "handler": hub_pull_tool, "mutates_state": True, @@ -428,19 +768,59 @@ async def hub_sync_tool( "name": "hub_search", "description": "Search for skills on the Hub by keyword.", "parameters": { - "query": "string (required) — search query", + "query": "string (required) \u2014 search query", + "federated": "boolean (optional) \u2014 search all backends in parallel (default: false)", }, "handler": hub_search_tool, }, + { + "name": "hub_federated_search", + "description": "Search for skills across ALL registered Hub backends in parallel.", + "parameters": { + "query": "string (required) \u2014 search query", + }, + "handler": hub_federated_search_tool, + }, { "name": "hub_sync", "description": "Preview or execute skill sync between local and Hub.", "parameters": { - "mode": "string (optional) — 'full' (default), 'push-only', or 'pull-only'", - "dry_run": "boolean (optional) — if true, only show plan (default: true)", + "mode": "string (optional) \u2014 'full' (default), 'push-only', or 'pull-only'", + "dry_run": "boolean (optional) \u2014 if true, only show plan (default: true)", }, "handler": hub_sync_tool, }, + { + "name": "hub_marketplace_browse", + "description": "Browse marketplace categories, featured and trending skills.", + "parameters": { + "category": "string (optional) — filter by category name", + "featured": "boolean (optional) — show featured/trending only (default: false)", + "trending": "boolean (optional) — show top skills by installs (default: false)", + "limit": "integer (optional) — max results (default: 10)", + }, + "handler": hub_marketplace_browse_tool, + }, + { + "name": "hub_marketplace_install", + "description": "Install a skill from the marketplace.", + "parameters": { + "name": "string (required) — skill name or repo_id", + }, + "handler": hub_marketplace_install_tool, + "mutates_state": True, + }, + { + "name": "hub_contribute", + "description": "Submit a skill to the community or check contribution status.", + "parameters": { + "skill_name": "string — name of the skill", + "action": "string (optional) — 'prepare', 'submit', 'status', or 'list'", + "hub_type": "string (optional) — target hub backend (default: 'github')", + }, + "handler": hub_contribute_tool, + "mutates_state": True, + }, ] diff --git a/tests/test_context_disclosure.py b/tests/test_context_disclosure.py index 6a2fe791..a3348a30 100644 --- a/tests/test_context_disclosure.py +++ b/tests/test_context_disclosure.py @@ -248,7 +248,7 @@ def test_config_tools_are_core_and_writes_are_not() -> None: def _desktop_definitions() -> list[dict]: from leapflow.skills.semantic_schema import semantic_tool_to_openai - from leapflow.skills.tool_executor import ToolDefinition + from leapflow.skills.tool_types import ToolDefinition defs = [] for name in ("observe_ui", "click", "list_apps"): diff --git a/tests/test_safety_and_policy.py b/tests/test_safety_and_policy.py index 0f32b643..913f3a88 100644 --- a/tests/test_safety_and_policy.py +++ b/tests/test_safety_and_policy.py @@ -17,7 +17,7 @@ default_rules, ) from leapflow.skills.sandbox import SandboxedNamespace -from leapflow.skills.tool_executor import ToolCall +from leapflow.skills.tool_types import ToolCall # ═══════════════════════════════════════════════════════════════════ diff --git a/tests/test_semantic_schema.py b/tests/test_semantic_schema.py index a69f1892..5dc49168 100644 --- a/tests/test_semantic_schema.py +++ b/tests/test_semantic_schema.py @@ -12,9 +12,12 @@ semantic_requires_approval, semantic_tool_to_openai, ) +from leapflow.skills.tool_types import ( + ToolCall, + ToolDefinition, +) from leapflow.skills.tool_executor import ( ExecutionToolset, - ToolDefinition, build_execution_toolset, ) from leapflow.plugins.tool_plugins.desktop_semantic import ( @@ -295,7 +298,7 @@ async def _noop(params: dict) -> dict: async def test_execution_toolset_unknown_tool_fails_cleanly() -> None: - from leapflow.skills.tool_executor import ToolCall + from leapflow.skills.tool_types import ToolCall toolset = build_execution_toolset(object(), perception=object()) result = await toolset.dispatch(ToolCall(name="nope", params={})) diff --git a/tests/test_slash_command_router.py b/tests/test_slash_command_router.py index 3434e4aa..40549217 100644 --- a/tests/test_slash_command_router.py +++ b/tests/test_slash_command_router.py @@ -181,7 +181,7 @@ def test_tools_payload_groups_desktop_tools_when_perception_online() -> None: from leapflow.cli.commands.slash_handlers import build_tool_payload from leapflow.skills.semantic_schema import semantic_tool_to_openai - from leapflow.skills.tool_executor import ToolDefinition + from leapflow.skills.tool_types import ToolDefinition from leapflow.plugins import get_registry _tool_reg = get_registry()