Skip to content
This repository was archived by the owner on Nov 10, 2025. It is now read-only.
Merged
Show file tree
Hide file tree
Changes from 8 commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
686dc25
feat: remove Embedchain adapter implementation
greysonlalonde Sep 12, 2025
9d6526c
feat: add CrewAI RAG adapter
greysonlalonde Sep 12, 2025
9f1e287
feat: update all search tools to use CrewAI RAG adapter
greysonlalonde Sep 12, 2025
4f83b69
test: update tests for CrewAI RAG adapter
greysonlalonde Sep 12, 2025
437f67d
chore: update dependencies and remove embedchain
greysonlalonde Sep 12, 2025
3340cbc
fix: improve CrewAI RAG adapter and fix Python 3.10 compatibility
greysonlalonde Sep 12, 2025
dd6c383
fix: convert CSV columns metadata to string for ChromaDB compatibility
greysonlalonde Sep 12, 2025
ea537fc
fix: sanitize metadata for ChromaDB compatibility
greysonlalonde Sep 15, 2025
6ac9d45
fix: integrate chunking and add similarity threshold to CrewAI RAG ad…
greysonlalonde Sep 17, 2025
0056a0b
feat: add PDF loader for RAG system
greysonlalonde Sep 17, 2025
d9ce123
fix: handle XML encoding issues in loader
greysonlalonde Sep 17, 2025
ae6d88f
fix: update RagTool config import for TYPE_CHECKING
greysonlalonde Sep 17, 2025
a73a646
chore: update crewai dependency to latest plugin-rag-factory branch
greysonlalonde Sep 17, 2025
1699396
chore: update PDF loader install instructions to use uv
greysonlalonde Sep 17, 2025
09b4518
fix: prevent double spaces in text chunker when separator is space
greysonlalonde Sep 17, 2025
6dd251e
feat: add YouTube video and channel loaders for RAG system
greysonlalonde Sep 17, 2025
4cd69cc
chore: add youtube-transcript-api dependency
greysonlalonde Sep 17, 2025
c5937fe
chore: update crewai dependency to main branch
greysonlalonde Sep 17, 2025
d2f81a8
chore: update uv.lock with refreshed crewai dependency
greysonlalonde Sep 17, 2025
76ba4b7
feat: add configurable similarity threshold and limit parameters to R…
greysonlalonde Sep 18, 2025
6d16abc
fix: properly sanitize YouTube URLs to prevent domain spoofing
greysonlalonde Sep 18, 2025
1bbe3e2
test: update search tool tests to include similarity threshold and li…
greysonlalonde Sep 18, 2025
7e6634d
feat: add loaders for GitHub, Docs Site, MySQL, and PostgreSQL
greysonlalonde Sep 18, 2025
cc1a3a2
chore: update crewai dependency to 0.186.1
greysonlalonde Sep 18, 2025
62c68cc
feat: add embedding_model config support to RagTool
greysonlalonde Sep 18, 2025
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
173 changes: 173 additions & 0 deletions crewai_tools/adapters/crewai_rag_adapter.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,173 @@
"""Adapter for CrewAI's native RAG system."""

from typing import Any, TypedDict, TypeAlias
from typing_extensions import Unpack
from pathlib import Path

from pydantic import Field
from crewai.rag.config.utils import get_rag_client
from crewai.rag.types import BaseRecord, SearchResult
from crewai.rag.core.base_client import BaseClient

from crewai_tools.tools.rag.rag_tool import Adapter
from crewai_tools.rag.data_types import DataType
from crewai_tools.rag.misc import sanitize_metadata_for_chromadb

ContentItem: TypeAlias = str | Path | dict[str, Any]

class AddDocumentParams(TypedDict, total=False):
"""Parameters for adding documents to the RAG system."""
data_type: DataType
metadata: dict[str, Any]
website: str
url: str
file_path: str | Path
github_url: str
youtube_url: str
directory_path: str | Path


class CrewAIRagAdapter(Adapter):
"""Adapter that uses CrewAI's native RAG system instead of embedchain."""

collection_name: str = "default"
summarize: bool = False
client: BaseClient = Field(default_factory=get_rag_client)

def model_post_init(self, __context: Any) -> None:
"""Initialize the CrewAI RAG client after model initialization."""
self.client.get_or_create_collection(collection_name=self.collection_name)

def query(self, question: str) -> str:
"""Query the knowledge base with a question.

Args:
question: The question to ask

Returns:
Relevant content from the knowledge base
"""
results: list[SearchResult] = self.client.search(
collection_name=self.collection_name,
query=question,
limit=5
)

if not results:
return "No relevant content found."

contents: list[str] = []
for result in results:
content: str = result.get("content", "")
if content:
contents.append(content)

return "\n\n".join(contents)

def add(self, *args: ContentItem, **kwargs: Unpack[AddDocumentParams]) -> None:
"""Add content to the knowledge base.

This method handles various input types and converts them to documents
for the vector database. It supports the data_type parameter for
compatibility with existing tools.

Args:
*args: Content items to add (strings, paths, or document dicts)
**kwargs: Additional parameters including data_type, metadata, etc.
"""
from crewai_tools.rag.data_types import DataTypes, DataType
from crewai_tools.rag.source_content import SourceContent
from crewai_tools.rag.base_loader import LoaderResult
import os

documents: list[BaseRecord] = []
data_type: DataType | None = kwargs.get("data_type")
base_metadata: dict[str, Any] = kwargs.get("metadata", {})

for arg in args:
source_ref: str
if isinstance(arg, dict):
source_ref = str(arg.get("source", arg.get("content", "")))
else:
source_ref = str(arg)

if not data_type:
data_type = DataTypes.from_content(source_ref)

if data_type == DataType.DIRECTORY:
if not os.path.isdir(source_ref):
raise ValueError(f"Directory does not exist: {source_ref}")

# Define binary and non-text file extensions to skip
binary_extensions = {'.pyc', '.pyo', '.png', '.jpg', '.jpeg', '.gif',
'.bmp', '.ico', '.svg', '.webp', '.pdf', '.zip',
'.tar', '.gz', '.bz2', '.7z', '.rar', '.exe',
'.dll', '.so', '.dylib', '.bin', '.dat', '.db',
'.sqlite', '.class', '.jar', '.war', '.ear'}

for root, dirs, files in os.walk(source_ref):
dirs[:] = [d for d in dirs if not d.startswith('.')]

for filename in files:
if filename.startswith('.'):
continue

# Skip binary files based on extension
file_ext = os.path.splitext(filename)[1].lower()
if file_ext in binary_extensions:
continue

# Skip __pycache__ directories
if '__pycache__' in root:
continue

file_path: str = os.path.join(root, filename)
try:
file_data_type: DataType = DataTypes.from_content(file_path)
file_loader = file_data_type.get_loader()

file_source = SourceContent(file_path)
file_result: LoaderResult = file_loader.load(file_source)

file_metadata: dict[str, Any] = base_metadata.copy()
file_metadata.update(file_result.metadata)
file_metadata["data_type"] = str(file_data_type)
file_metadata["directory_source"] = source_ref
file_metadata["file_path"] = file_path

if isinstance(arg, dict):
file_metadata.update(arg.get("metadata", {}))

documents.append({
"doc_id": file_result.doc_id,
"content": file_result.content,
"metadata": sanitize_metadata_for_chromadb(file_metadata)
})
except Exception:
# Silently skip files that can't be processed
continue
else:
metadata: dict[str, Any] = base_metadata.copy()

loader = data_type.get_loader()

source_content = SourceContent(source_ref)
loader_result: LoaderResult = loader.load(source_content)

metadata.update(loader_result.metadata)
metadata["data_type"] = str(data_type)

if isinstance(arg, dict):
metadata.update(arg.get("metadata", {}))

documents.append({
"doc_id": loader_result.doc_id,
"content": loader_result.content,
"metadata": sanitize_metadata_for_chromadb(metadata)
})

if documents:
self.client.add_documents(
collection_name=self.collection_name,
documents=documents
)
34 changes: 0 additions & 34 deletions crewai_tools/adapters/embedchain_adapter.py

This file was deleted.

41 changes: 0 additions & 41 deletions crewai_tools/adapters/pdf_embedchain_adapter.py

This file was deleted.

2 changes: 2 additions & 0 deletions crewai_tools/rag/data_types.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,8 @@ class DataType(str, Enum):
# Web types
WEBSITE = "website"
DOCS_SITE = "docs_site"
YOUTUBE_VIDEO = "youtube_video"
YOUTUBE_CHANNEL = "youtube_channel"

# Raw types
TEXT = "text"
Expand Down
25 changes: 25 additions & 0 deletions crewai_tools/rag/misc.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,29 @@
import hashlib
from typing import Any

def compute_sha256(content: str) -> str:
return hashlib.sha256(content.encode("utf-8")).hexdigest()

def sanitize_metadata_for_chromadb(metadata: dict[str, Any]) -> dict[str, Any]:
"""Sanitize metadata to ensure ChromaDB compatibility.

ChromaDB only accepts str, int, float, or bool values in metadata.
This function converts other types to strings.

Args:
metadata: Dictionary of metadata to sanitize

Returns:
Sanitized metadata dictionary with only ChromaDB-compatible types
"""
sanitized = {}
for key, value in metadata.items():
if isinstance(value, (str, int, float, bool)) or value is None:
sanitized[key] = value
elif isinstance(value, (list, tuple)):
# Convert lists/tuples to pipe-separated strings
sanitized[key] = " | ".join(str(v) for v in value)
else:
# Convert other types to string
sanitized[key] = str(value)
return sanitized
Original file line number Diff line number Diff line change
@@ -1,14 +1,10 @@
from typing import Any, Optional, Type

try:
from embedchain.models.data_type import DataType
EMBEDCHAIN_AVAILABLE = True
except ImportError:
EMBEDCHAIN_AVAILABLE = False

from pydantic import BaseModel, Field

from ..rag.rag_tool import RagTool
from crewai_tools.rag.data_types import DataType


class FixedCodeDocsSearchToolSchema(BaseModel):
Expand Down Expand Up @@ -42,8 +38,6 @@ def __init__(self, docs_url: Optional[str] = None, **kwargs):
self._generate_description()

def add(self, docs_url: str) -> None:
if not EMBEDCHAIN_AVAILABLE:
raise ImportError("embedchain is not installed. Please install it with `pip install crewai-tools[embedchain]`")
super().add(docs_url, data_type=DataType.DOCS_SITE)

def _run(
Expand Down
8 changes: 1 addition & 7 deletions crewai_tools/tools/csv_search_tool/csv_search_tool.py
Original file line number Diff line number Diff line change
@@ -1,14 +1,10 @@
from typing import Optional, Type

try:
from embedchain.models.data_type import DataType
EMBEDCHAIN_AVAILABLE = True
except ImportError:
EMBEDCHAIN_AVAILABLE = False

from pydantic import BaseModel, Field

from ..rag.rag_tool import RagTool
from crewai_tools.rag.data_types import DataType


class FixedCSVSearchToolSchema(BaseModel):
Expand Down Expand Up @@ -42,8 +38,6 @@ def __init__(self, csv: Optional[str] = None, **kwargs):
self._generate_description()

def add(self, csv: str) -> None:
if not EMBEDCHAIN_AVAILABLE:
raise ImportError("embedchain is not installed. Please install it with `pip install crewai-tools[embedchain]`")
super().add(csv, data_type=DataType.CSV)

def _run(
Expand Down
14 changes: 2 additions & 12 deletions crewai_tools/tools/directory_search_tool/directory_search_tool.py
Original file line number Diff line number Diff line change
@@ -1,14 +1,9 @@
from typing import Optional, Type

try:
from embedchain.loaders.directory_loader import DirectoryLoader
EMBEDCHAIN_AVAILABLE = True
except ImportError:
EMBEDCHAIN_AVAILABLE = False

from pydantic import BaseModel, Field

from ..rag.rag_tool import RagTool
from crewai_tools.rag.data_types import DataType


class FixedDirectorySearchToolSchema(BaseModel):
Expand All @@ -34,8 +29,6 @@ class DirectorySearchTool(RagTool):
args_schema: Type[BaseModel] = DirectorySearchToolSchema

def __init__(self, directory: Optional[str] = None, **kwargs):
if not EMBEDCHAIN_AVAILABLE:
raise ImportError("embedchain is not installed. Please install it with `pip install crewai-tools[embedchain]`")
super().__init__(**kwargs)
if directory is not None:
self.add(directory)
Expand All @@ -44,10 +37,7 @@ def __init__(self, directory: Optional[str] = None, **kwargs):
self._generate_description()

def add(self, directory: str) -> None:
super().add(
directory,
loader=DirectoryLoader(config=dict(recursive=True)),
)
super().add(directory, data_type=DataType.DIRECTORY)

def _run(
self,
Expand Down
Loading
Loading