mirror of
https://github.com/basicmachines-co/basic-memory
synced 2026-06-21 13:47:35 +00:00
feat: Optimize directory operations for 10-100x performance improvement (#350)
Signed-off-by: phernandez <paul@basicmachines.co> Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -176,6 +176,66 @@ class EntityRepository(Repository[Entity]):
|
||||
entity = await self._handle_permalink_conflict(entity, session)
|
||||
return entity
|
||||
|
||||
async def get_distinct_directories(self) -> List[str]:
|
||||
"""Extract unique directory paths from file_path column.
|
||||
|
||||
Optimized method for getting directory structure without loading full entities
|
||||
or relationships. Returns a sorted list of unique directory paths.
|
||||
|
||||
Returns:
|
||||
List of unique directory paths (e.g., ["notes", "notes/meetings", "specs"])
|
||||
"""
|
||||
# Query only file_path column, no entity objects or relationships
|
||||
query = select(Entity.file_path).distinct()
|
||||
query = self._add_project_filter(query)
|
||||
|
||||
# Execute with use_query_options=False to skip eager loading
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
file_paths = [row for row in result.scalars().all()]
|
||||
|
||||
# Parse file paths to extract unique directories
|
||||
directories = set()
|
||||
for file_path in file_paths:
|
||||
parts = [p for p in file_path.split("/") if p]
|
||||
# Add all parent directories (exclude filename which is the last part)
|
||||
for i in range(len(parts) - 1):
|
||||
dir_path = "/".join(parts[: i + 1])
|
||||
directories.add(dir_path)
|
||||
|
||||
return sorted(directories)
|
||||
|
||||
async def find_by_directory_prefix(self, directory_prefix: str) -> Sequence[Entity]:
|
||||
"""Find entities whose file_path starts with the given directory prefix.
|
||||
|
||||
Optimized method for listing directory contents without loading all entities.
|
||||
Uses SQL LIKE pattern matching to filter entities by directory path.
|
||||
|
||||
Args:
|
||||
directory_prefix: Directory path prefix (e.g., "docs", "docs/guides")
|
||||
Empty string returns all entities (root directory)
|
||||
|
||||
Returns:
|
||||
Sequence of entities in the specified directory and subdirectories
|
||||
"""
|
||||
# Build SQL LIKE pattern
|
||||
if directory_prefix == "" or directory_prefix == "/":
|
||||
# Root directory - return all entities
|
||||
return await self.find_all()
|
||||
|
||||
# Remove leading/trailing slashes for consistency
|
||||
directory_prefix = directory_prefix.strip("/")
|
||||
|
||||
# Query entities with file_path starting with prefix
|
||||
# Pattern matches "prefix/" to ensure we get files IN the directory,
|
||||
# not just files whose names start with the prefix
|
||||
pattern = f"{directory_prefix}/%"
|
||||
|
||||
query = self.select().where(Entity.file_path.like(pattern))
|
||||
|
||||
# Skip eager loading - we only need basic entity fields for directory trees
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return list(result.scalars().all())
|
||||
|
||||
async def _handle_permalink_conflict(self, entity: Entity, session: AsyncSession) -> Entity:
|
||||
"""Handle permalink conflicts by generating a unique permalink."""
|
||||
base_permalink = entity.permalink
|
||||
|
||||
@@ -152,12 +152,25 @@ class Repository[T: Base]:
|
||||
# Add project filter if applicable
|
||||
return self._add_project_filter(query)
|
||||
|
||||
async def find_all(self, skip: int = 0, limit: Optional[int] = None) -> Sequence[T]:
|
||||
"""Fetch records from the database with pagination."""
|
||||
async def find_all(
|
||||
self, skip: int = 0, limit: Optional[int] = None, use_load_options: bool = True
|
||||
) -> Sequence[T]:
|
||||
"""Fetch records from the database with pagination.
|
||||
|
||||
Args:
|
||||
skip: Number of records to skip
|
||||
limit: Maximum number of records to return
|
||||
use_load_options: Whether to apply eager loading options (default: True)
|
||||
"""
|
||||
logger.debug(f"Finding all {self.Model.__name__} (skip={skip}, limit={limit})")
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
query = select(self.Model).offset(skip).options(*self.get_load_options())
|
||||
query = select(self.Model).offset(skip)
|
||||
|
||||
# Only apply load options if requested
|
||||
if use_load_options:
|
||||
query = query.options(*self.get_load_options())
|
||||
|
||||
# Add project filter if applicable
|
||||
query = self._add_project_filter(query)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user