mirror of
https://github.com/basicmachines-co/basic-memory
synced 2026-06-21 13:47:35 +00:00
Compare commits
43 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 24dc9a2931 | |||
| 85c63e5a7a | |||
| f227ef6a86 | |||
| 897b1edaa4 | |||
| 0c12a39a98 | |||
| efbc758325 | |||
| a0f20eb102 | |||
| 78673d8e51 | |||
| 126c0495c0 | |||
| 4a43d7df4a | |||
| c462faf046 | |||
| 70bb10be1d | |||
| fbf9045d78 | |||
| 1094210c52 | |||
| 391feb639f | |||
| a920a9ff29 | |||
| 05efe8701c | |||
| 0eaf30bb06 | |||
| 0818bda565 | |||
| 6f99d2e551 | |||
| 73d940e064 | |||
| c3678a11d2 | |||
| 203d684c24 | |||
| a872220924 | |||
| 7d763a66ff | |||
| b5d4fb559c | |||
| 830775276d | |||
| ed894fc3ed | |||
| 704338edcf | |||
| 0ca02a7ebe | |||
| 28cc5225a7 | |||
| 9b7bbc7116 | |||
| 138c283d6c | |||
| 7a8954c37e | |||
| 10c7c19c03 | |||
| fb5e9e1d77 | |||
| 66b91b2847 | |||
| b004565df9 | |||
| a258b73e1d | |||
| 9a845f2906 | |||
| 6517e9845f | |||
| 1af05392ee | |||
| cad7019c89 |
@@ -15,10 +15,16 @@ Create a stable release using the automated justfile target with comprehensive v
|
||||
You are an expert release manager for the Basic Memory project. When the user runs `/release`, execute the following steps:
|
||||
|
||||
### Step 1: Pre-flight Validation
|
||||
1. Verify version format matches `v\d+\.\d+\.\d+` pattern
|
||||
2. Check current git status for uncommitted changes
|
||||
3. Verify we're on the `main` branch
|
||||
4. Confirm no existing tag with this version
|
||||
|
||||
#### Version Check
|
||||
1. Check current version in `src/basic_memory/__init__.py`
|
||||
2. Verify new version format matches `v\d+\.\d+\.\d+` pattern
|
||||
3. Confirm version is higher than current version
|
||||
|
||||
#### Git Status
|
||||
1. Check current git status for uncommitted changes
|
||||
2. Verify we're on the `main` branch
|
||||
3. Confirm no existing tag with this version
|
||||
|
||||
#### Documentation Validation
|
||||
1. **Changelog Check**
|
||||
@@ -39,19 +45,83 @@ The justfile target handles:
|
||||
- ✅ Version update in `src/basic_memory/__init__.py`
|
||||
- ✅ Automatic commit with proper message
|
||||
- ✅ Tag creation and pushing to GitHub
|
||||
- ✅ Release workflow trigger
|
||||
- ✅ Release workflow trigger (automatic on tag push)
|
||||
|
||||
The GitHub Actions workflow (`.github/workflows/release.yml`) then:
|
||||
- ✅ Builds the package using `uv build`
|
||||
- ✅ Creates GitHub release with auto-generated notes
|
||||
- ✅ Publishes to PyPI
|
||||
- ✅ Updates Homebrew formula (stable releases only)
|
||||
|
||||
### Step 3: Monitor Release Process
|
||||
1. Check that GitHub Actions workflow starts successfully
|
||||
2. Monitor workflow completion at: https://github.com/basicmachines-co/basic-memory/actions
|
||||
3. Verify PyPI publication
|
||||
4. Test installation: `uv tool install basic-memory`
|
||||
1. Verify tag push triggered the workflow (should start automatically within seconds)
|
||||
2. Monitor workflow progress at: https://github.com/basicmachines-co/basic-memory/actions
|
||||
3. Watch for successful completion of both jobs:
|
||||
- `release` - Builds package and publishes to PyPI
|
||||
- `homebrew` - Updates Homebrew formula (stable releases only)
|
||||
4. Check for any workflow failures and investigate logs if needed
|
||||
|
||||
### Step 4: Post-Release Validation
|
||||
1. Verify GitHub release is created automatically
|
||||
2. Check PyPI publication
|
||||
3. Validate release assets
|
||||
4. Update any post-release documentation
|
||||
|
||||
#### GitHub Release
|
||||
1. Verify GitHub release is created at: https://github.com/basicmachines-co/basic-memory/releases/tag/<version>
|
||||
2. Check that release notes are auto-generated from commits
|
||||
3. Validate release assets (`.whl` and `.tar.gz` files are attached)
|
||||
|
||||
#### PyPI Publication
|
||||
1. Verify package published at: https://pypi.org/project/basic-memory/<version>/
|
||||
2. Test installation: `uv tool install basic-memory`
|
||||
3. Verify installed version: `basic-memory --version`
|
||||
|
||||
#### Homebrew Formula (Stable Releases Only)
|
||||
1. Check formula update at: https://github.com/basicmachines-co/homebrew-basic-memory
|
||||
2. Verify formula version matches release
|
||||
3. Test Homebrew installation: `brew install basicmachines-co/basic-memory/basic-memory`
|
||||
|
||||
#### Website Updates
|
||||
|
||||
**1. basicmachines.co** (`/Users/drew/code/basicmachines.co`)
|
||||
- **Goal**: Update version number displayed on the homepage
|
||||
- **Location**: Search for "Basic Memory v0." in the codebase to find version displays
|
||||
- **What to update**:
|
||||
- Hero section heading that shows "Basic Memory v{VERSION}"
|
||||
- "What's New in v{VERSION}" section heading
|
||||
- Feature highlights array (look for array of features with title/description)
|
||||
- **Process**:
|
||||
1. Pull latest from GitHub: `git pull origin main`
|
||||
2. Create release branch: `git checkout -b release/v{VERSION}`
|
||||
3. Search codebase for current version number (e.g., "v0.16.1")
|
||||
4. Update version numbers to new release version
|
||||
5. Update feature highlights with 3-5 key features from this release (extract from CHANGELOG.md)
|
||||
6. Commit changes: `git commit -m "chore: update to v{VERSION}"`
|
||||
7. Push branch: `git push origin release/v{VERSION}`
|
||||
- **Deploy**: Follow deployment process for basicmachines.co
|
||||
|
||||
**2. docs.basicmemory.com** (`/Users/drew/code/docs.basicmemory.com`)
|
||||
- **Goal**: Add new release notes section to the latest-releases page
|
||||
- **File**: `src/pages/latest-releases.mdx`
|
||||
- **What to do**:
|
||||
1. Pull latest from GitHub: `git pull origin main`
|
||||
2. Create release branch: `git checkout -b release/v{VERSION}`
|
||||
3. Read the existing file to understand the format and structure
|
||||
4. Read `/Users/drew/code/basic-memory/CHANGELOG.md` to get release content
|
||||
5. Add new release section **at the top** (after MDX imports, before other releases)
|
||||
6. Follow the existing pattern:
|
||||
- Heading: `## [v{VERSION}](github-link) — YYYY-MM-DD`
|
||||
- Focus statement if applicable
|
||||
- `<Info>` block with highlights (3-5 key items)
|
||||
- Sections for Features, Bug Fixes, Breaking Changes, etc.
|
||||
- Link to full changelog at the end
|
||||
- Separator `---` between releases
|
||||
7. Commit changes: `git commit -m "docs: add v{VERSION} release notes"`
|
||||
8. Push branch: `git push origin release/v{VERSION}`
|
||||
- **Source content**: Extract and format sections from CHANGELOG.md for this version
|
||||
- **Deploy**: Follow deployment process for docs.basicmemory.com
|
||||
|
||||
**4. Announce Release**
|
||||
- Post to Discord community if significant changes
|
||||
- Update social media if major release
|
||||
- Notify users via appropriate channels
|
||||
|
||||
## Pre-conditions Check
|
||||
Before starting, verify:
|
||||
@@ -74,13 +144,18 @@ Before starting, verify:
|
||||
🏷️ Tag: v0.13.2
|
||||
📋 GitHub Release: https://github.com/basicmachines-co/basic-memory/releases/tag/v0.13.2
|
||||
📦 PyPI: https://pypi.org/project/basic-memory/0.13.2/
|
||||
🍺 Homebrew: https://github.com/basicmachines-co/homebrew-basic-memory
|
||||
🚀 GitHub Actions: Completed
|
||||
|
||||
Install with:
|
||||
uv tool install basic-memory
|
||||
Install with pip/uv:
|
||||
uv tool install basic-memory
|
||||
|
||||
Install with Homebrew:
|
||||
brew install basicmachines-co/basic-memory/basic-memory
|
||||
|
||||
Users can now upgrade:
|
||||
uv tool upgrade basic-memory
|
||||
uv tool upgrade basic-memory
|
||||
brew upgrade basic-memory
|
||||
```
|
||||
|
||||
## Context
|
||||
@@ -89,4 +164,6 @@ uv tool upgrade basic-memory
|
||||
- Uses the automated justfile target for consistency
|
||||
- Version is automatically updated in `__init__.py`
|
||||
- Triggers automated GitHub release with changelog
|
||||
- Leverages uv-dynamic-versioning for package version management
|
||||
- Package is published to PyPI for `pip` and `uv` users
|
||||
- Homebrew formula is automatically updated for stable releases
|
||||
- Supports multiple installation methods (uv, pip, Homebrew)
|
||||
+16
-20
@@ -1,17 +1,19 @@
|
||||
---
|
||||
allowed-tools: mcp__basic-memory__write_note, mcp__basic-memory__read_note, mcp__basic-memory__search_notes, mcp__basic-memory__edit_note, Task
|
||||
argument-hint: [create|status|implement|review] [spec-name]
|
||||
allowed-tools: mcp__basic-memory__write_note, mcp__basic-memory__read_note, mcp__basic-memory__search_notes, mcp__basic-memory__edit_note
|
||||
argument-hint: [create|status|show|review] [spec-name]
|
||||
description: Manage specifications in our development process
|
||||
---
|
||||
|
||||
## Context
|
||||
|
||||
You are managing specifications using our specification-driven development process defined in @docs/specs/SPEC-001.md.
|
||||
Specifications are managed in the Basic Memory "specs" project. All specs live in a centralized location accessible across all repositories via MCP tools.
|
||||
|
||||
See SPEC-1 and SPEC-2 in the "specs" project for the full specification-driven development process.
|
||||
|
||||
Available commands:
|
||||
- `create [name]` - Create new specification
|
||||
- `status` - Show all spec statuses
|
||||
- `implement [spec-name]` - Hand spec to appropriate agent
|
||||
- `show [spec-name]` - Read a specific spec
|
||||
- `review [spec-name]` - Review implementation against spec
|
||||
|
||||
## Your task
|
||||
@@ -19,23 +21,19 @@ Available commands:
|
||||
Execute the spec command: `/spec $ARGUMENTS`
|
||||
|
||||
### If command is "create":
|
||||
1. Get next SPEC number by searching existing specs
|
||||
2. Create new spec using template from @docs/specs/Slash\ Commands\ Reference.md
|
||||
3. Place in `/specs` folder with title "SPEC-XXX: [name]"
|
||||
1. Get next SPEC number by searching existing specs in "specs" project
|
||||
2. Create new spec using template from SPEC-2
|
||||
3. Use mcp__basic-memory__write_note with project="specs"
|
||||
4. Include standard sections: Why, What, How, How to Evaluate
|
||||
|
||||
### If command is "status":
|
||||
1. Search all notes in `/specs` folder
|
||||
2. Display table with spec number, title, and status
|
||||
3. Show any dependencies or assigned agents
|
||||
1. Use mcp__basic-memory__search_notes with project="specs"
|
||||
2. Display table with spec number, title, and progress
|
||||
3. Show completion status from checkboxes in content
|
||||
|
||||
### If command is "implement":
|
||||
1. Read the specified spec
|
||||
2. Determine appropriate agent based on content:
|
||||
- Frontend/UI → vue-developer
|
||||
- Architecture/system → system-architect
|
||||
- Backend/API → python-developer
|
||||
3. Launch Task tool with appropriate agent and spec context
|
||||
### If command is "show":
|
||||
1. Use mcp__basic-memory__read_note with project="specs"
|
||||
2. Display the full spec content
|
||||
|
||||
### If command is "review":
|
||||
1. Read the specified spec and its "How to Evaluate" section
|
||||
@@ -49,7 +47,5 @@ Execute the spec command: `/spec $ARGUMENTS`
|
||||
- **Architecture compliance** - Component isolation, state management patterns
|
||||
- **Documentation completeness** - Implementation matches specification
|
||||
3. Provide honest, accurate assessment - do not overstate completeness
|
||||
4. Document findings and update spec with review results
|
||||
4. Document findings and update spec with review results using mcp__basic-memory__edit_note
|
||||
5. If gaps found, clearly identify what still needs to be implemented/tested
|
||||
|
||||
Use the agent definitions from @docs/specs/Agent\ Definitions.md for implementation handoffs.
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"enabledPlugins": {
|
||||
"basic-memory@basicmachines": true
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
# Basic Memory Environment Variables Example
|
||||
# Copy this file to .env and customize as needed
|
||||
# Note: .env files are gitignored and should never be committed
|
||||
|
||||
# ============================================================================
|
||||
# PostgreSQL Test Database Configuration
|
||||
# ============================================================================
|
||||
# These variables allow you to override the default test database credentials
|
||||
# Default values match docker-compose-postgres.yml for local development
|
||||
#
|
||||
# Only needed if you want to use different credentials or a remote test database
|
||||
# By default, tests use: postgresql://basic_memory_user:dev_password@localhost:5433/basic_memory_test
|
||||
|
||||
# Full PostgreSQL test database URL (used by tests and migrations)
|
||||
# POSTGRES_TEST_URL=postgresql+asyncpg://basic_memory_user:dev_password@localhost:5433/basic_memory_test
|
||||
|
||||
# Individual components (used by justfile postgres-reset command)
|
||||
# POSTGRES_USER=basic_memory_user
|
||||
# POSTGRES_TEST_DB=basic_memory_test
|
||||
|
||||
# ============================================================================
|
||||
# Production Database Configuration
|
||||
# ============================================================================
|
||||
# For production use, set these in your deployment environment
|
||||
# DO NOT use the test credentials above in production!
|
||||
|
||||
# BASIC_MEMORY_DATABASE_BACKEND=postgres # or "sqlite"
|
||||
# BASIC_MEMORY_DATABASE_URL=postgresql+asyncpg://user:password@host:port/database
|
||||
@@ -13,7 +13,8 @@ on:
|
||||
branches: [ "main" ]
|
||||
|
||||
jobs:
|
||||
test:
|
||||
test-sqlite:
|
||||
name: Test SQLite (${{ matrix.os }}, Python ${{ matrix.python-version }})
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
@@ -64,7 +65,49 @@ jobs:
|
||||
run: |
|
||||
just lint
|
||||
|
||||
- name: Run tests
|
||||
- name: Run tests (SQLite)
|
||||
run: |
|
||||
uv pip install pytest pytest-cov
|
||||
just test
|
||||
just test-sqlite
|
||||
|
||||
test-postgres:
|
||||
name: Test Postgres (Python ${{ matrix.python-version }})
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: [ "3.12", "3.13" ]
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
# Note: No services section needed - testcontainers handles Postgres in Docker
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
submodules: true
|
||||
|
||||
- name: Set up Python ${{ matrix.python-version }}
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
cache: 'pip'
|
||||
|
||||
- name: Install uv
|
||||
run: |
|
||||
pip install uv
|
||||
|
||||
- name: Install just
|
||||
run: |
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://just.systems/install.sh | bash -s -- --to /usr/local/bin
|
||||
|
||||
- name: Create virtual env
|
||||
run: |
|
||||
uv venv
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
uv pip install -e .[dev]
|
||||
|
||||
- name: Run tests (Postgres via testcontainers)
|
||||
run: |
|
||||
uv pip install pytest pytest-cov
|
||||
just test-postgres
|
||||
+2
-1
@@ -52,4 +52,5 @@ ENV/
|
||||
|
||||
# claude action
|
||||
claude-output
|
||||
**/.claude/settings.local.json
|
||||
**/.claude/settings.local.json
|
||||
.mcp.json
|
||||
|
||||
+115
@@ -1,5 +1,120 @@
|
||||
# CHANGELOG
|
||||
|
||||
## v0.16.3 (2025-12-20)
|
||||
|
||||
### Features
|
||||
|
||||
- **#439**: Add PostgreSQL database backend support
|
||||
([`fb5e9e1`](https://github.com/basicmachines-co/basic-memory/commit/fb5e9e1))
|
||||
- Full PostgreSQL/Neon database support as alternative to SQLite
|
||||
- Async connection pooling with asyncpg
|
||||
- Alembic migrations support for both backends
|
||||
- Configurable via `BASIC_MEMORY_DATABASE_BACKEND` environment variable
|
||||
|
||||
- **#441**: Implement API v2 with ID-based endpoints (Phase 1)
|
||||
([`28cc522`](https://github.com/basicmachines-co/basic-memory/commit/28cc522))
|
||||
- New ID-based API endpoints for improved performance
|
||||
- Foundation for future API enhancements
|
||||
- Backward compatible with existing endpoints
|
||||
|
||||
- Add project_id to Relation and Observation for efficient project-scoped queries
|
||||
([`a920a9f`](https://github.com/basicmachines-co/basic-memory/commit/a920a9f))
|
||||
- Enables faster queries in multi-project environments
|
||||
- Improved database schema for cloud deployments
|
||||
|
||||
- Add bulk insert with ON CONFLICT handling for relations
|
||||
([`0818bda`](https://github.com/basicmachines-co/basic-memory/commit/0818bda))
|
||||
- Faster relation creation during sync operations
|
||||
- Handles duplicate relations gracefully
|
||||
|
||||
### Performance
|
||||
|
||||
- Lightweight permalink resolution to avoid eager loading
|
||||
([`6f99d2e`](https://github.com/basicmachines-co/basic-memory/commit/6f99d2e))
|
||||
- Reduces database queries during entity lookups
|
||||
- Improved response times for read operations
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- **#464**: Pin FastMCP to 2.12.3 to fix MCP tools visibility
|
||||
([`f227ef6`](https://github.com/basicmachines-co/basic-memory/commit/f227ef6))
|
||||
- Fixes issue where MCP tools were not visible to Claude
|
||||
- Reverts to last known working FastMCP version
|
||||
|
||||
- **#458**: Reduce watch service CPU usage by increasing reload interval
|
||||
([`897b1ed`](https://github.com/basicmachines-co/basic-memory/commit/897b1ed))
|
||||
- Lowers CPU usage during file watching
|
||||
- More efficient resource utilization
|
||||
|
||||
- **#456**: Await background sync task cancellation in lifespan shutdown
|
||||
([`efbc758`](https://github.com/basicmachines-co/basic-memory/commit/efbc758))
|
||||
- Prevents hanging on shutdown
|
||||
- Clean async task cleanup
|
||||
|
||||
- **#434**: Respect --project flag in background sync
|
||||
([`70bb10b`](https://github.com/basicmachines-co/basic-memory/commit/70bb10b))
|
||||
- Background sync now correctly uses specified project
|
||||
- Fixes multi-project sync issues
|
||||
|
||||
- **#446**: Fix observation parsing and permalink limits
|
||||
([`73d940e`](https://github.com/basicmachines-co/basic-memory/commit/73d940e))
|
||||
- Handles edge cases in observation content
|
||||
- Prevents permalink truncation issues
|
||||
|
||||
- **#424**: Handle periods in kebab_filenames mode
|
||||
([`b004565`](https://github.com/basicmachines-co/basic-memory/commit/b004565))
|
||||
- Fixes filename handling for files with multiple periods
|
||||
- Improved kebab-case conversion
|
||||
|
||||
- Fix Postgres/Neon connection settings and search index dedupe
|
||||
([`b5d4fb5`](https://github.com/basicmachines-co/basic-memory/commit/b5d4fb5))
|
||||
- Optimized connection pooling for Postgres
|
||||
- Prevents duplicate search index entries
|
||||
|
||||
### Testing & CI
|
||||
|
||||
- Replace py-pglite with testcontainers for Postgres testing
|
||||
([`c462faf`](https://github.com/basicmachines-co/basic-memory/commit/c462faf))
|
||||
- More reliable Postgres testing infrastructure
|
||||
- Uses Docker-based test containers
|
||||
|
||||
- Add PostgreSQL testing to GitHub Actions workflow
|
||||
([`66b91b2`](https://github.com/basicmachines-co/basic-memory/commit/66b91b2))
|
||||
- CI now tests both SQLite and PostgreSQL backends
|
||||
- Ensures cross-database compatibility
|
||||
|
||||
- **#416**: Add integration test for read_note with underscored folders
|
||||
([`0c12a39`](https://github.com/basicmachines-co/basic-memory/commit/0c12a39))
|
||||
- Verifies folder name handling edge cases
|
||||
|
||||
### Internal
|
||||
|
||||
- Cloud compatibility fixes and performance improvements (#454)
|
||||
- Remove logfire instrumentation for cleaner production deployments
|
||||
- Truncate content_stems to fix Postgres 8KB index row limit
|
||||
|
||||
## v0.16.2 (2025-11-16)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
- **#429**: Use platform-native path separators in config.json
|
||||
([`6517e98`](https://github.com/basicmachines-co/basic-memory/commit/6517e98))
|
||||
- Fixes config.json path separator issues on Windows
|
||||
- Uses os.path.join for platform-native path construction
|
||||
- Ensures consistent path handling across platforms
|
||||
|
||||
- **#427**: Add rclone installation checks for Windows bisync commands
|
||||
([`1af0539`](https://github.com/basicmachines-co/basic-memory/commit/1af0539))
|
||||
- Validates rclone installation before running bisync commands
|
||||
- Provides clear error messages when rclone is not installed
|
||||
- Improves user experience on Windows
|
||||
|
||||
- **#421**: Main project always recreated on project list command
|
||||
([`cad7019`](https://github.com/basicmachines-co/basic-memory/commit/cad7019))
|
||||
- Fixes issue where main project was recreated unnecessarily
|
||||
- Improves project list command reliability
|
||||
- Reduces unnecessary file system operations
|
||||
|
||||
## v0.16.1 (2025-11-11)
|
||||
|
||||
### Bug Fixes
|
||||
|
||||
@@ -15,10 +15,14 @@ See the [README.md](README.md) file for a project overview.
|
||||
### Build and Test Commands
|
||||
|
||||
- Install: `just install` or `pip install -e ".[dev]"`
|
||||
- Run all tests (with coverage): `just test` - Runs both unit and integration tests with unified coverage
|
||||
- Run unit tests only: `just test-unit` - Fast, no coverage
|
||||
- Run integration tests only: `just test-int` - Fast, no coverage
|
||||
- Generate HTML coverage: `just coverage` - Opens in browser
|
||||
- Run all tests (SQLite + Postgres): `just test`
|
||||
- Run all tests against SQLite: `just test-sqlite`
|
||||
- Run all tests against Postgres: `just test-postgres` (uses testcontainers)
|
||||
- Run unit tests (SQLite): `just test-unit-sqlite`
|
||||
- Run unit tests (Postgres): `just test-unit-postgres`
|
||||
- Run integration tests (SQLite): `just test-int-sqlite`
|
||||
- Run integration tests (Postgres): `just test-int-postgres`
|
||||
- Generate HTML coverage: `just coverage`
|
||||
- Single test: `pytest tests/path/to/test_file.py::test_function_name`
|
||||
- Run benchmarks: `pytest test-int/test_sync_performance_benchmark.py -v -m "benchmark and not slow"`
|
||||
- Lint: `just lint` or `ruff check . --fix`
|
||||
@@ -30,6 +34,8 @@ See the [README.md](README.md) file for a project overview.
|
||||
|
||||
**Note:** Project requires Python 3.12+ (uses type parameter syntax and `type` aliases introduced in 3.12)
|
||||
|
||||
**Postgres Testing:** Uses [testcontainers](https://testcontainers-python.readthedocs.io/) which automatically spins up a Postgres instance in Docker. No manual database setup required - just have Docker running.
|
||||
|
||||
### Test Structure
|
||||
|
||||
- `tests/` - Unit tests for individual components (mocked, fast)
|
||||
@@ -76,8 +82,10 @@ See the [README.md](README.md) file for a project overview.
|
||||
- SQLite is used for indexing and full text search, files are source of truth
|
||||
- Testing uses pytest with asyncio support (strict mode)
|
||||
- Unit tests (`tests/`) use mocks when necessary; integration tests (`test-int/`) use real implementations
|
||||
- Test database uses in-memory SQLite
|
||||
- Each test runs in a standalone environment with in-memory SQLite and tmp_file directory
|
||||
- By default, tests run against SQLite (fast, no Docker needed)
|
||||
- Set `BASIC_MEMORY_TEST_POSTGRES=1` to run against Postgres (uses testcontainers - Docker required)
|
||||
- Each test runs in a standalone environment with isolated database and tmp_path directory
|
||||
- CI runs SQLite and Postgres tests in parallel for faster feedback
|
||||
- Performance benchmarks are in `test-int/test_sync_performance_benchmark.py`
|
||||
- Use pytest markers: `@pytest.mark.benchmark` for benchmarks, `@pytest.mark.slow` for slow tests
|
||||
|
||||
@@ -229,6 +237,11 @@ of using AI just for code generation, we've developed a true collaborative workf
|
||||
This approach has allowed us to tackle more complex challenges and build a more robust system than either humans or AI
|
||||
could achieve independently.
|
||||
|
||||
**Problem-Solving Guidance:**
|
||||
- If a solution isn't working after reasonable effort, suggest alternative approaches
|
||||
- Don't persist with a problematic library or pattern when better alternatives exist
|
||||
- Example: When py-pglite caused cascading test failures, switching to testcontainers-postgres was the right call
|
||||
|
||||
## GitHub Integration
|
||||
|
||||
Basic Memory has taken AI-Human collaboration to the next level by integrating Claude directly into the development workflow through GitHub:
|
||||
@@ -264,5 +277,6 @@ With GitHub integration, the development workflow includes:
|
||||
2. **Contribution tracking** - All of Claude's contributions are properly attributed in the Git history
|
||||
3. **Branch management** - Claude can create feature branches for implementations
|
||||
4. **Documentation maintenance** - Claude can keep documentation updated as the code evolves
|
||||
5. **Code Commits**: ALWAYS sign off commits with `git commit -s`
|
||||
|
||||
This level of integration represents a new paradigm in AI-human collaboration, where the AI assistant becomes a full-fledged team member rather than just a tool for generating code snippets.
|
||||
|
||||
@@ -433,6 +433,91 @@ See the [Documentation](https://memory.basicmachines.co/) for more info, includi
|
||||
- [Managing multiple Projects](https://docs.basicmemory.com/guides/cli-reference/#project)
|
||||
- [Importing data from OpenAI/Claude Projects](https://docs.basicmemory.com/guides/cli-reference/#import)
|
||||
|
||||
## Logging
|
||||
|
||||
Basic Memory uses [Loguru](https://github.com/Delgan/loguru) for logging. The logging behavior varies by entry point:
|
||||
|
||||
| Entry Point | Default Behavior | Use Case |
|
||||
|-------------|------------------|----------|
|
||||
| CLI commands | File only | Prevents log output from interfering with command output |
|
||||
| MCP server | File only | Stdout would corrupt the JSON-RPC protocol |
|
||||
| API server | File (local) or stdout (cloud) | Docker/cloud deployments use stdout |
|
||||
|
||||
**Log file location:** `~/.basic-memory/basic-memory.log` (10MB rotation, 10 days retention)
|
||||
|
||||
### Environment Variables
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `BASIC_MEMORY_LOG_LEVEL` | `INFO` | Log level: DEBUG, INFO, WARNING, ERROR |
|
||||
| `BASIC_MEMORY_CLOUD_MODE` | `false` | When `true`, API logs to stdout with structured context |
|
||||
| `BASIC_MEMORY_ENV` | `dev` | Set to `test` for test mode (stderr only) |
|
||||
|
||||
### Examples
|
||||
|
||||
```bash
|
||||
# Enable debug logging
|
||||
BASIC_MEMORY_LOG_LEVEL=DEBUG basic-memory sync
|
||||
|
||||
# View logs
|
||||
tail -f ~/.basic-memory/basic-memory.log
|
||||
|
||||
# Cloud/Docker mode (stdout logging with structured context)
|
||||
BASIC_MEMORY_CLOUD_MODE=true uvicorn basic_memory.api.app:app
|
||||
```
|
||||
|
||||
## Development
|
||||
|
||||
### Running Tests
|
||||
|
||||
Basic Memory supports dual database backends (SQLite and Postgres). By default, tests run against SQLite. Set `BASIC_MEMORY_TEST_POSTGRES=1` to run against Postgres (uses testcontainers - Docker required).
|
||||
|
||||
**Quick Start:**
|
||||
```bash
|
||||
# Run all tests against SQLite (default, fast)
|
||||
just test-sqlite
|
||||
|
||||
# Run all tests against Postgres (uses testcontainers)
|
||||
just test-postgres
|
||||
|
||||
# Run both SQLite and Postgres tests
|
||||
just test
|
||||
```
|
||||
|
||||
**Available Test Commands:**
|
||||
|
||||
- `just test` - Run all tests against both SQLite and Postgres
|
||||
- `just test-sqlite` - Run all tests against SQLite (fast, no Docker needed)
|
||||
- `just test-postgres` - Run all tests against Postgres (uses testcontainers)
|
||||
- `just test-unit-sqlite` - Run unit tests against SQLite
|
||||
- `just test-unit-postgres` - Run unit tests against Postgres
|
||||
- `just test-int-sqlite` - Run integration tests against SQLite
|
||||
- `just test-int-postgres` - Run integration tests against Postgres
|
||||
- `just test-windows` - Run Windows-specific tests (auto-skips on other platforms)
|
||||
- `just test-benchmark` - Run performance benchmark tests
|
||||
|
||||
**Postgres Testing:**
|
||||
|
||||
Postgres tests use [testcontainers](https://testcontainers-python.readthedocs.io/) which automatically spins up a Postgres instance in Docker. No manual database setup required - just have Docker running.
|
||||
|
||||
**Test Markers:**
|
||||
|
||||
Tests use pytest markers for selective execution:
|
||||
- `windows` - Windows-specific database optimizations
|
||||
- `benchmark` - Performance tests (excluded from default runs)
|
||||
|
||||
**Other Development Commands:**
|
||||
```bash
|
||||
just install # Install with dev dependencies
|
||||
just lint # Run linting checks
|
||||
just typecheck # Run type checking
|
||||
just format # Format code with ruff
|
||||
just check # Run all quality checks
|
||||
just migration "msg" # Create database migration
|
||||
```
|
||||
|
||||
See the [justfile](justfile) for the complete list of development commands.
|
||||
|
||||
## License
|
||||
|
||||
AGPL-3.0
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
# Docker Compose configuration for Basic Memory with PostgreSQL
|
||||
# Use this for local development and testing with Postgres backend
|
||||
#
|
||||
# Usage:
|
||||
# docker-compose -f docker-compose-postgres.yml up -d
|
||||
# docker-compose -f docker-compose-postgres.yml down
|
||||
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:17
|
||||
container_name: basic-memory-postgres
|
||||
environment:
|
||||
# Local development/test credentials - NOT for production
|
||||
# These values are referenced by tests and justfile commands
|
||||
POSTGRES_DB: basic_memory
|
||||
POSTGRES_USER: basic_memory_user
|
||||
POSTGRES_PASSWORD: dev_password # Simple password for local testing only
|
||||
ports:
|
||||
- "5433:5432"
|
||||
volumes:
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U basic_memory_user -d basic_memory"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
# Named volume for Postgres data
|
||||
postgres_data:
|
||||
driver: local
|
||||
|
||||
# Named volume for persistent configuration
|
||||
# Database will be stored in Postgres, not in this volume
|
||||
basic-memory-config:
|
||||
driver: local
|
||||
|
||||
# Network configuration (optional)
|
||||
# networks:
|
||||
# basic-memory-net:
|
||||
# driver: bridge
|
||||
@@ -7,16 +7,85 @@ install:
|
||||
@echo ""
|
||||
@echo "💡 Remember to activate the virtual environment by running: source .venv/bin/activate"
|
||||
|
||||
# Run unit tests only (fast, no coverage)
|
||||
test-unit:
|
||||
uv run pytest -p pytest_mock -v --no-cov -n auto tests
|
||||
# ==============================================================================
|
||||
# DATABASE BACKEND TESTING
|
||||
# ==============================================================================
|
||||
# Basic Memory supports dual database backends (SQLite and Postgres).
|
||||
# By default, tests run against SQLite (fast, no dependencies).
|
||||
# Set BASIC_MEMORY_TEST_POSTGRES=1 to run against Postgres (uses testcontainers).
|
||||
#
|
||||
# Quick Start:
|
||||
# just test # Run all tests against SQLite (default)
|
||||
# just test-sqlite # Run all tests against SQLite
|
||||
# just test-postgres # Run all tests against Postgres (testcontainers)
|
||||
# just test-unit-sqlite # Run unit tests against SQLite
|
||||
# just test-unit-postgres # Run unit tests against Postgres
|
||||
# just test-int-sqlite # Run integration tests against SQLite
|
||||
# just test-int-postgres # Run integration tests against Postgres
|
||||
#
|
||||
# CI runs both in parallel for faster feedback.
|
||||
# ==============================================================================
|
||||
|
||||
# Run integration tests only (fast, no coverage)
|
||||
test-int:
|
||||
uv run pytest -p pytest_mock -v --no-cov -n auto test-int
|
||||
# Run all tests against SQLite and Postgres
|
||||
test: test-sqlite test-postgres
|
||||
|
||||
# Run all tests with unified coverage report
|
||||
test: test-unit test-int
|
||||
# Run all tests against SQLite
|
||||
test-sqlite: test-unit-sqlite test-int-sqlite
|
||||
|
||||
# Run all tests against Postgres (uses testcontainers)
|
||||
test-postgres: test-unit-postgres test-int-postgres
|
||||
|
||||
# Run unit tests against SQLite
|
||||
test-unit-sqlite:
|
||||
uv run pytest -p pytest_mock -v --no-cov tests
|
||||
|
||||
# Run unit tests against Postgres
|
||||
test-unit-postgres:
|
||||
BASIC_MEMORY_TEST_POSTGRES=1 uv run pytest -p pytest_mock -v --no-cov tests
|
||||
|
||||
# Run integration tests against SQLite
|
||||
test-int-sqlite:
|
||||
uv run pytest -p pytest_mock -v --no-cov test-int
|
||||
|
||||
# Run integration tests against Postgres
|
||||
# Note: Uses timeout due to FastMCP Client + asyncpg cleanup hang (tests pass, process hangs on exit)
|
||||
# See: https://github.com/jlowin/fastmcp/issues/1311
|
||||
test-int-postgres:
|
||||
timeout --signal=KILL 600 bash -c 'BASIC_MEMORY_TEST_POSTGRES=1 uv run pytest -p pytest_mock -v --no-cov test-int' || test $? -eq 137
|
||||
|
||||
# Reset Postgres test database (drops and recreates schema)
|
||||
# Useful when Alembic migration state gets out of sync during development
|
||||
# Uses credentials from docker-compose-postgres.yml
|
||||
postgres-reset:
|
||||
docker exec basic-memory-postgres psql -U ${POSTGRES_USER:-basic_memory_user} -d ${POSTGRES_TEST_DB:-basic_memory_test} -c "DROP SCHEMA public CASCADE; CREATE SCHEMA public;"
|
||||
@echo "✅ Postgres test database reset"
|
||||
|
||||
# Run Alembic migrations manually against Postgres test database
|
||||
# Useful for debugging migration issues
|
||||
# Uses credentials from docker-compose-postgres.yml (can override with env vars)
|
||||
postgres-migrate:
|
||||
@cd src/basic_memory/alembic && \
|
||||
BASIC_MEMORY_DATABASE_BACKEND=postgres \
|
||||
BASIC_MEMORY_DATABASE_URL=${POSTGRES_TEST_URL:-postgresql+asyncpg://basic_memory_user:dev_password@localhost:5433/basic_memory_test} \
|
||||
uv run alembic upgrade head
|
||||
@echo "✅ Migrations applied to Postgres test database"
|
||||
|
||||
# Run Windows-specific tests only (only works on Windows platform)
|
||||
# These tests verify Windows-specific database optimizations (locking mode, NullPool)
|
||||
# Will be skipped automatically on non-Windows platforms
|
||||
test-windows:
|
||||
uv run pytest -p pytest_mock -v --no-cov -m windows tests test-int
|
||||
|
||||
# Run benchmark tests only (performance testing)
|
||||
# These are slow tests that measure sync performance with various file counts
|
||||
# Excluded from default test runs to keep CI fast
|
||||
test-benchmark:
|
||||
uv run pytest -p pytest_mock -v --no-cov -m benchmark tests test-int
|
||||
|
||||
# Run all tests including Windows, Postgres, and Benchmarks (for CI/comprehensive testing)
|
||||
# Use this before releasing to ensure everything works across all backends and platforms
|
||||
test-all:
|
||||
uv run pytest -p pytest_mock -v --no-cov tests test-int
|
||||
|
||||
# Generate HTML coverage report
|
||||
coverage:
|
||||
|
||||
+10
-4
@@ -15,7 +15,6 @@ dependencies = [
|
||||
"aiosqlite>=0.20.0",
|
||||
"greenlet>=3.1.1",
|
||||
"pydantic[email,timezone]>=2.10.3",
|
||||
"icecream>=2.1.3",
|
||||
"mcp>=1.2.0",
|
||||
"pydantic-settings>=2.6.1",
|
||||
"loguru>=0.7.3",
|
||||
@@ -30,12 +29,15 @@ dependencies = [
|
||||
"alembic>=1.14.1",
|
||||
"pillow>=11.1.0",
|
||||
"pybars3>=0.9.7",
|
||||
"fastmcp>=2.10.2",
|
||||
"fastmcp==2.12.3", # Pinned - 2.14.x breaks MCP tools visibility (issue #463)
|
||||
"pyjwt>=2.10.1",
|
||||
"python-dotenv>=1.1.0",
|
||||
"pytest-aio>=1.9.0",
|
||||
"aiofiles>=24.1.0", # Async file I/O
|
||||
"logfire>=0.73.0", # Optional observability (disabled by default via config)
|
||||
"aiofiles>=24.1.0", # Optional observability (disabled by default via config)
|
||||
"asyncpg>=0.30.0",
|
||||
"nest-asyncio>=1.6.0", # For Alembic migrations with Postgres
|
||||
"pytest-asyncio>=1.2.0",
|
||||
"psycopg==3.3.1",
|
||||
]
|
||||
|
||||
|
||||
@@ -61,6 +63,8 @@ asyncio_default_fixture_loop_scope = "function"
|
||||
markers = [
|
||||
"benchmark: Performance benchmark tests (deselect with '-m \"not benchmark\"')",
|
||||
"slow: Slow-running tests (deselect with '-m \"not slow\"')",
|
||||
"postgres: Tests that run against Postgres backend (deselect with '-m \"not postgres\"')",
|
||||
"windows: Windows-specific tests (deselect with '-m \"not windows\"')",
|
||||
]
|
||||
|
||||
[tool.ruff]
|
||||
@@ -78,6 +82,8 @@ dev = [
|
||||
"pytest-xdist>=3.0.0",
|
||||
"ruff>=0.1.6",
|
||||
"freezegun>=1.5.5",
|
||||
"testcontainers[postgres]>=4.0.0",
|
||||
"psycopg>=3.2.0",
|
||||
]
|
||||
|
||||
[tool.hatch.version]
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""basic-memory - Local-first knowledge management combining Zettelkasten with knowledge graphs"""
|
||||
|
||||
# Package version - updated by release automation
|
||||
__version__ = "0.16.1"
|
||||
__version__ = "0.16.3"
|
||||
|
||||
# API version for FastAPI - independent of package version
|
||||
__api_version__ = "v0"
|
||||
|
||||
+104
-22
@@ -1,10 +1,21 @@
|
||||
"""Alembic environment configuration."""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
from logging.config import fileConfig
|
||||
|
||||
from sqlalchemy import engine_from_config
|
||||
from sqlalchemy import pool
|
||||
# Allow nested event loops (needed for pytest-asyncio and other async contexts)
|
||||
# Note: nest_asyncio doesn't work with uvloop, so we handle that case separately
|
||||
try:
|
||||
import nest_asyncio
|
||||
|
||||
nest_asyncio.apply()
|
||||
except (ImportError, ValueError):
|
||||
# nest_asyncio not available or can't patch this loop type (e.g., uvloop)
|
||||
pass
|
||||
|
||||
from sqlalchemy import engine_from_config, pool
|
||||
from sqlalchemy.ext.asyncio import AsyncEngine, create_async_engine
|
||||
|
||||
from alembic import context
|
||||
|
||||
@@ -20,12 +31,22 @@ from basic_memory.models import Base # noqa: E402
|
||||
# access to the values within the .ini file in use.
|
||||
config = context.config
|
||||
|
||||
# Load app config - this will read environment variables (BASIC_MEMORY_DATABASE_BACKEND, etc.)
|
||||
# due to Pydantic's env_prefix="BASIC_MEMORY_" setting
|
||||
app_config = ConfigManager().config
|
||||
# Set the SQLAlchemy URL from our app config
|
||||
sqlalchemy_url = f"sqlite:///{app_config.database_path}"
|
||||
config.set_main_option("sqlalchemy.url", sqlalchemy_url)
|
||||
|
||||
# print(f"Using SQLAlchemy URL: {sqlalchemy_url}")
|
||||
# Set the SQLAlchemy URL based on database backend configuration
|
||||
# If the URL is already set in config (e.g., from run_migrations), use that
|
||||
# Otherwise, get it from app config
|
||||
# Note: alembic.ini has a placeholder URL "driver://user:pass@localhost/dbname" that we need to override
|
||||
current_url = config.get_main_option("sqlalchemy.url")
|
||||
if not current_url or current_url == "driver://user:pass@localhost/dbname":
|
||||
from basic_memory.db import DatabaseType
|
||||
|
||||
sqlalchemy_url = DatabaseType.get_db_url(
|
||||
app_config.database_path, DatabaseType.FILESYSTEM, app_config
|
||||
)
|
||||
config.set_main_option("sqlalchemy.url", sqlalchemy_url)
|
||||
|
||||
# Interpret the config file for Python logging.
|
||||
if config.config_file_name is not None:
|
||||
@@ -69,28 +90,89 @@ def run_migrations_offline() -> None:
|
||||
context.run_migrations()
|
||||
|
||||
|
||||
def do_run_migrations(connection):
|
||||
"""Execute migrations with the given connection."""
|
||||
context.configure(
|
||||
connection=connection,
|
||||
target_metadata=target_metadata,
|
||||
include_object=include_object,
|
||||
render_as_batch=True,
|
||||
compare_type=True,
|
||||
)
|
||||
with context.begin_transaction():
|
||||
context.run_migrations()
|
||||
|
||||
|
||||
async def run_async_migrations(connectable):
|
||||
"""Run migrations asynchronously with AsyncEngine."""
|
||||
async with connectable.connect() as connection:
|
||||
await connection.run_sync(do_run_migrations)
|
||||
await connectable.dispose()
|
||||
|
||||
|
||||
def run_migrations_online() -> None:
|
||||
"""Run migrations in 'online' mode.
|
||||
|
||||
In this scenario we need to create an Engine
|
||||
and associate a connection with the context.
|
||||
Supports both sync engines (SQLite) and async engines (PostgreSQL with asyncpg).
|
||||
"""
|
||||
connectable = engine_from_config(
|
||||
config.get_section(config.config_ini_section, {}),
|
||||
prefix="sqlalchemy.",
|
||||
poolclass=pool.NullPool,
|
||||
)
|
||||
# Check if a connection/engine was provided (e.g., from run_migrations)
|
||||
connectable = context.config.attributes.get("connection", None)
|
||||
|
||||
with connectable.connect() as connection:
|
||||
context.configure(
|
||||
connection=connection,
|
||||
target_metadata=target_metadata,
|
||||
include_object=include_object,
|
||||
render_as_batch=True,
|
||||
)
|
||||
if connectable is None:
|
||||
# No connection provided, create engine from config
|
||||
url = context.config.get_main_option("sqlalchemy.url")
|
||||
|
||||
with context.begin_transaction():
|
||||
context.run_migrations()
|
||||
# Check if it's an async URL (sqlite+aiosqlite or postgresql+asyncpg)
|
||||
if url and ("+asyncpg" in url or "+aiosqlite" in url):
|
||||
# Create async engine for asyncpg or aiosqlite
|
||||
connectable = create_async_engine(
|
||||
url,
|
||||
poolclass=pool.NullPool,
|
||||
future=True,
|
||||
)
|
||||
else:
|
||||
# Create sync engine for regular sqlite or postgresql
|
||||
connectable = engine_from_config(
|
||||
context.config.get_section(context.config.config_ini_section, {}),
|
||||
prefix="sqlalchemy.",
|
||||
poolclass=pool.NullPool,
|
||||
)
|
||||
|
||||
# Handle async engines (PostgreSQL with asyncpg)
|
||||
if isinstance(connectable, AsyncEngine):
|
||||
# Try to run async migrations
|
||||
# nest_asyncio allows asyncio.run() from within event loops, but doesn't work with uvloop
|
||||
try:
|
||||
asyncio.run(run_async_migrations(connectable))
|
||||
except RuntimeError as e:
|
||||
if "cannot be called from a running event loop" in str(e):
|
||||
# We're in a running event loop (likely uvloop) - need to use a different approach
|
||||
# Create a new thread to run the async migrations
|
||||
import concurrent.futures
|
||||
|
||||
def run_in_thread():
|
||||
"""Run async migrations in a new event loop in a separate thread."""
|
||||
new_loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(new_loop)
|
||||
try:
|
||||
new_loop.run_until_complete(run_async_migrations(connectable))
|
||||
finally:
|
||||
new_loop.close()
|
||||
|
||||
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||
future = executor.submit(run_in_thread)
|
||||
future.result() # Wait for completion and re-raise any exceptions
|
||||
else:
|
||||
raise
|
||||
else:
|
||||
# Handle sync engines (SQLite) or sync connections
|
||||
if hasattr(connectable, "connect"):
|
||||
# It's an engine, get a connection
|
||||
with connectable.connect() as connection:
|
||||
do_run_migrations(connection)
|
||||
else:
|
||||
# It's already a connection
|
||||
do_run_migrations(connectable)
|
||||
|
||||
|
||||
if context.is_offline_mode():
|
||||
|
||||
+131
@@ -0,0 +1,131 @@
|
||||
"""Add Postgres full-text search support with tsvector and GIN indexes
|
||||
|
||||
Revision ID: 314f1ea54dc4
|
||||
Revises: e7e1f4367280
|
||||
Create Date: 2025-11-15 18:05:01.025405
|
||||
|
||||
"""
|
||||
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "314f1ea54dc4"
|
||||
down_revision: Union[str, None] = "e7e1f4367280"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Add PostgreSQL full-text search support.
|
||||
|
||||
This migration:
|
||||
1. Creates search_index table for Postgres (SQLite uses FTS5 virtual table)
|
||||
2. Adds generated tsvector column for full-text search
|
||||
3. Creates GIN index on the tsvector column for fast text queries
|
||||
4. Creates GIN index on metadata JSONB column for fast containment queries
|
||||
|
||||
Note: These changes only apply to Postgres. SQLite continues to use FTS5 virtual tables.
|
||||
"""
|
||||
# Check if we're using Postgres
|
||||
connection = op.get_bind()
|
||||
if connection.dialect.name == "postgresql":
|
||||
# Create search_index table for Postgres
|
||||
# For SQLite, this is a FTS5 virtual table created elsewhere
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
|
||||
op.create_table(
|
||||
"search_index",
|
||||
sa.Column("id", sa.Integer(), nullable=False), # Entity IDs are integers
|
||||
sa.Column("project_id", sa.Integer(), nullable=False), # Multi-tenant isolation
|
||||
sa.Column("title", sa.Text(), nullable=True),
|
||||
sa.Column("content_stems", sa.Text(), nullable=True),
|
||||
sa.Column("content_snippet", sa.Text(), nullable=True),
|
||||
sa.Column("permalink", sa.String(), nullable=True), # Nullable for non-markdown files
|
||||
sa.Column("file_path", sa.String(), nullable=True),
|
||||
sa.Column("type", sa.String(), nullable=True),
|
||||
sa.Column("from_id", sa.Integer(), nullable=True), # Relation IDs are integers
|
||||
sa.Column("to_id", sa.Integer(), nullable=True), # Relation IDs are integers
|
||||
sa.Column("relation_type", sa.String(), nullable=True),
|
||||
sa.Column("entity_id", sa.Integer(), nullable=True), # Entity IDs are integers
|
||||
sa.Column("category", sa.String(), nullable=True),
|
||||
sa.Column("metadata", JSONB(), nullable=True), # Use JSONB for Postgres
|
||||
sa.Column("created_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.Column("updated_at", sa.DateTime(timezone=True), nullable=True),
|
||||
sa.PrimaryKeyConstraint(
|
||||
"id", "type", "project_id"
|
||||
), # Composite key: id can repeat across types
|
||||
sa.ForeignKeyConstraint(
|
||||
["project_id"],
|
||||
["project.id"],
|
||||
name="fk_search_index_project_id",
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
if_not_exists=True,
|
||||
)
|
||||
|
||||
# Create index on project_id for efficient multi-tenant queries
|
||||
op.create_index(
|
||||
"ix_search_index_project_id",
|
||||
"search_index",
|
||||
["project_id"],
|
||||
unique=False,
|
||||
)
|
||||
|
||||
# Create unique partial index on permalink for markdown files
|
||||
# Non-markdown files don't have permalinks, so we use a partial index
|
||||
op.execute("""
|
||||
CREATE UNIQUE INDEX uix_search_index_permalink_project
|
||||
ON search_index (permalink, project_id)
|
||||
WHERE permalink IS NOT NULL
|
||||
""")
|
||||
|
||||
# Add tsvector column as a GENERATED ALWAYS column
|
||||
# This automatically updates when title or content_stems change
|
||||
op.execute("""
|
||||
ALTER TABLE search_index
|
||||
ADD COLUMN textsearchable_index_col tsvector
|
||||
GENERATED ALWAYS AS (
|
||||
to_tsvector('english',
|
||||
coalesce(title, '') || ' ' ||
|
||||
coalesce(content_stems, '')
|
||||
)
|
||||
) STORED
|
||||
""")
|
||||
|
||||
# Create GIN index on tsvector column for fast full-text search
|
||||
op.create_index(
|
||||
"idx_search_index_fts",
|
||||
"search_index",
|
||||
["textsearchable_index_col"],
|
||||
unique=False,
|
||||
postgresql_using="gin",
|
||||
)
|
||||
|
||||
# Create GIN index on metadata JSONB for fast containment queries
|
||||
# Using jsonb_path_ops for smaller index size and better performance
|
||||
op.execute("""
|
||||
CREATE INDEX idx_search_index_metadata_gin
|
||||
ON search_index
|
||||
USING GIN (metadata jsonb_path_ops)
|
||||
""")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Remove PostgreSQL full-text search support."""
|
||||
connection = op.get_bind()
|
||||
if connection.dialect.name == "postgresql":
|
||||
# Drop indexes first
|
||||
op.execute("DROP INDEX IF EXISTS idx_search_index_metadata_gin")
|
||||
op.drop_index("idx_search_index_fts", table_name="search_index")
|
||||
op.execute("DROP INDEX IF EXISTS uix_search_index_permalink_project")
|
||||
op.drop_index("ix_search_index_project_id", table_name="search_index")
|
||||
|
||||
# Drop the generated column
|
||||
op.execute("ALTER TABLE search_index DROP COLUMN IF EXISTS textsearchable_index_col")
|
||||
|
||||
# Drop the search_index table
|
||||
op.drop_table("search_index")
|
||||
@@ -21,6 +21,12 @@ depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
def upgrade() -> None:
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
|
||||
# SQLite FTS5 virtual table handling is SQLite-specific
|
||||
# For Postgres, search_index is a regular table managed by ORM
|
||||
connection = op.get_bind()
|
||||
is_sqlite = connection.dialect.name == "sqlite"
|
||||
|
||||
op.create_table(
|
||||
"project",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
@@ -55,7 +61,9 @@ def upgrade() -> None:
|
||||
batch_op.add_column(sa.Column("project_id", sa.Integer(), nullable=False))
|
||||
batch_op.drop_index(
|
||||
"uix_entity_permalink",
|
||||
sqlite_where=sa.text("content_type = 'text/markdown' AND permalink IS NOT NULL"),
|
||||
sqlite_where=sa.text("content_type = 'text/markdown' AND permalink IS NOT NULL")
|
||||
if is_sqlite
|
||||
else None,
|
||||
)
|
||||
batch_op.drop_index("ix_entity_file_path")
|
||||
batch_op.create_index(batch_op.f("ix_entity_file_path"), ["file_path"], unique=False)
|
||||
@@ -67,12 +75,16 @@ def upgrade() -> None:
|
||||
"uix_entity_permalink_project",
|
||||
["permalink", "project_id"],
|
||||
unique=True,
|
||||
sqlite_where=sa.text("content_type = 'text/markdown' AND permalink IS NOT NULL"),
|
||||
sqlite_where=sa.text("content_type = 'text/markdown' AND permalink IS NOT NULL")
|
||||
if is_sqlite
|
||||
else None,
|
||||
)
|
||||
batch_op.create_foreign_key("fk_entity_project_id", "project", ["project_id"], ["id"])
|
||||
|
||||
# drop the search index table. it will be recreated
|
||||
op.drop_table("search_index")
|
||||
# Only drop for SQLite - Postgres uses regular table managed by ORM
|
||||
if is_sqlite:
|
||||
op.drop_table("search_index")
|
||||
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
@@ -25,43 +25,51 @@ def upgrade() -> None:
|
||||
The UNIQUE constraint prevents multiple projects from having is_default=FALSE,
|
||||
which breaks project creation when the service sets is_default=False.
|
||||
|
||||
Since SQLite doesn't support dropping specific constraints easily, we'll
|
||||
recreate the table without the problematic constraint.
|
||||
SQLite: Recreate the table without the constraint (no ALTER TABLE support)
|
||||
Postgres: Use ALTER TABLE to drop the constraint directly
|
||||
"""
|
||||
# For SQLite, we need to recreate the table without the UNIQUE constraint
|
||||
# Create a new table without the UNIQUE constraint on is_default
|
||||
op.create_table(
|
||||
"project_new",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("description", sa.Text(), nullable=True),
|
||||
sa.Column("permalink", sa.String(), nullable=False),
|
||||
sa.Column("path", sa.String(), nullable=False),
|
||||
sa.Column("is_active", sa.Boolean(), nullable=False),
|
||||
sa.Column("is_default", sa.Boolean(), nullable=True), # No UNIQUE constraint!
|
||||
sa.Column("created_at", sa.DateTime(), nullable=False),
|
||||
sa.Column("updated_at", sa.DateTime(), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id"),
|
||||
sa.UniqueConstraint("name"),
|
||||
sa.UniqueConstraint("permalink"),
|
||||
)
|
||||
connection = op.get_bind()
|
||||
is_sqlite = connection.dialect.name == "sqlite"
|
||||
|
||||
# Copy data from old table to new table
|
||||
op.execute("INSERT INTO project_new SELECT * FROM project")
|
||||
if is_sqlite:
|
||||
# For SQLite, we need to recreate the table without the UNIQUE constraint
|
||||
# Create a new table without the UNIQUE constraint on is_default
|
||||
op.create_table(
|
||||
"project_new",
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("description", sa.Text(), nullable=True),
|
||||
sa.Column("permalink", sa.String(), nullable=False),
|
||||
sa.Column("path", sa.String(), nullable=False),
|
||||
sa.Column("is_active", sa.Boolean(), nullable=False),
|
||||
sa.Column("is_default", sa.Boolean(), nullable=True), # No UNIQUE constraint!
|
||||
sa.Column("created_at", sa.DateTime(), nullable=False),
|
||||
sa.Column("updated_at", sa.DateTime(), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id"),
|
||||
sa.UniqueConstraint("name"),
|
||||
sa.UniqueConstraint("permalink"),
|
||||
)
|
||||
|
||||
# Drop the old table
|
||||
op.drop_table("project")
|
||||
# Copy data from old table to new table
|
||||
op.execute("INSERT INTO project_new SELECT * FROM project")
|
||||
|
||||
# Rename the new table
|
||||
op.rename_table("project_new", "project")
|
||||
# Drop the old table
|
||||
op.drop_table("project")
|
||||
|
||||
# Recreate the indexes
|
||||
with op.batch_alter_table("project", schema=None) as batch_op:
|
||||
batch_op.create_index("ix_project_created_at", ["created_at"], unique=False)
|
||||
batch_op.create_index("ix_project_name", ["name"], unique=True)
|
||||
batch_op.create_index("ix_project_path", ["path"], unique=False)
|
||||
batch_op.create_index("ix_project_permalink", ["permalink"], unique=True)
|
||||
batch_op.create_index("ix_project_updated_at", ["updated_at"], unique=False)
|
||||
# Rename the new table
|
||||
op.rename_table("project_new", "project")
|
||||
|
||||
# Recreate the indexes
|
||||
with op.batch_alter_table("project", schema=None) as batch_op:
|
||||
batch_op.create_index("ix_project_created_at", ["created_at"], unique=False)
|
||||
batch_op.create_index("ix_project_name", ["name"], unique=True)
|
||||
batch_op.create_index("ix_project_path", ["path"], unique=False)
|
||||
batch_op.create_index("ix_project_permalink", ["permalink"], unique=True)
|
||||
batch_op.create_index("ix_project_updated_at", ["updated_at"], unique=False)
|
||||
else:
|
||||
# For Postgres, we can simply drop the constraint
|
||||
with op.batch_alter_table("project", schema=None) as batch_op:
|
||||
batch_op.drop_constraint("project_is_default_key", type_="unique")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Add cascade delete FK from search_index to entity
|
||||
|
||||
Revision ID: a2b3c4d5e6f7
|
||||
Revises: f8a9b2c3d4e5
|
||||
Create Date: 2025-12-02 07:00:00.000000
|
||||
|
||||
"""
|
||||
|
||||
from typing import Sequence, Union
|
||||
|
||||
from alembic import op
|
||||
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "a2b3c4d5e6f7"
|
||||
down_revision: Union[str, None] = "f8a9b2c3d4e5"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Add FK with CASCADE delete from search_index.entity_id to entity.id.
|
||||
|
||||
This migration is Postgres-only because:
|
||||
- SQLite uses FTS5 virtual tables which don't support foreign keys
|
||||
- The FK enables automatic cleanup of search_index entries when entities are deleted
|
||||
"""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
if dialect == "postgresql":
|
||||
# First, clean up any orphaned search_index entries where entity no longer exists
|
||||
op.execute("""
|
||||
DELETE FROM search_index
|
||||
WHERE entity_id IS NOT NULL
|
||||
AND entity_id NOT IN (SELECT id FROM entity)
|
||||
""")
|
||||
|
||||
# Add FK with CASCADE - nullable FK allows search_index entries without entity_id
|
||||
op.create_foreign_key(
|
||||
"fk_search_index_entity_id",
|
||||
"search_index",
|
||||
"entity",
|
||||
["entity_id"],
|
||||
["id"],
|
||||
ondelete="CASCADE",
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Remove the FK constraint."""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
if dialect == "postgresql":
|
||||
op.drop_constraint("fk_search_index_entity_id", "search_index", type_="foreignkey")
|
||||
@@ -21,6 +21,12 @@ depends_on: Union[str, Sequence[str], None] = None
|
||||
def upgrade() -> None:
|
||||
"""Upgrade database schema to use new search index with content_stems and content_snippet."""
|
||||
|
||||
# This migration is SQLite-specific (FTS5 virtual tables)
|
||||
# For Postgres, the search_index table is created via ORM models
|
||||
connection = op.get_bind()
|
||||
if connection.dialect.name != "sqlite":
|
||||
return
|
||||
|
||||
# First, drop the existing search_index table
|
||||
op.execute("DROP TABLE IF EXISTS search_index")
|
||||
|
||||
@@ -59,6 +65,13 @@ def upgrade() -> None:
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade database schema to use old search index."""
|
||||
|
||||
# This migration is SQLite-specific (FTS5 virtual tables)
|
||||
# For Postgres, the search_index table is managed via ORM models
|
||||
connection = op.get_bind()
|
||||
if connection.dialect.name != "sqlite":
|
||||
return
|
||||
|
||||
# Drop the updated search_index table
|
||||
op.execute("DROP TABLE IF EXISTS search_index")
|
||||
|
||||
|
||||
+199
@@ -0,0 +1,199 @@
|
||||
"""Add project_id to relation/observation and pg_trgm for fuzzy link resolution
|
||||
|
||||
Revision ID: f8a9b2c3d4e5
|
||||
Revises: 314f1ea54dc4
|
||||
Create Date: 2025-12-01 12:00:00.000000
|
||||
|
||||
"""
|
||||
|
||||
from typing import Sequence, Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "f8a9b2c3d4e5"
|
||||
down_revision: Union[str, None] = "314f1ea54dc4"
|
||||
branch_labels: Union[str, Sequence[str], None] = None
|
||||
depends_on: Union[str, Sequence[str], None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Add project_id to relation and observation tables, plus pg_trgm indexes.
|
||||
|
||||
This migration:
|
||||
1. Adds project_id column to relation and observation tables (denormalization)
|
||||
2. Backfills project_id from the associated entity
|
||||
3. Enables pg_trgm extension for trigram-based fuzzy matching (Postgres only)
|
||||
4. Creates GIN indexes on entity title and permalink for fast similarity searches
|
||||
5. Creates partial index on unresolved relations for efficient bulk resolution
|
||||
"""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Add project_id to relation table
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
# Step 1: Add project_id column as nullable first
|
||||
op.add_column("relation", sa.Column("project_id", sa.Integer(), nullable=True))
|
||||
|
||||
# Step 2: Backfill project_id from entity.project_id via from_id
|
||||
if dialect == "postgresql":
|
||||
op.execute("""
|
||||
UPDATE relation
|
||||
SET project_id = entity.project_id
|
||||
FROM entity
|
||||
WHERE relation.from_id = entity.id
|
||||
""")
|
||||
else:
|
||||
# SQLite syntax
|
||||
op.execute("""
|
||||
UPDATE relation
|
||||
SET project_id = (
|
||||
SELECT entity.project_id
|
||||
FROM entity
|
||||
WHERE entity.id = relation.from_id
|
||||
)
|
||||
""")
|
||||
|
||||
# Step 3: Make project_id NOT NULL and add foreign key
|
||||
if dialect == "postgresql":
|
||||
op.alter_column("relation", "project_id", nullable=False)
|
||||
op.create_foreign_key(
|
||||
"fk_relation_project_id",
|
||||
"relation",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
else:
|
||||
# SQLite requires batch operations for ALTER COLUMN
|
||||
with op.batch_alter_table("relation") as batch_op:
|
||||
batch_op.alter_column("project_id", nullable=False)
|
||||
batch_op.create_foreign_key(
|
||||
"fk_relation_project_id",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
|
||||
# Step 4: Create index on relation.project_id
|
||||
op.create_index("ix_relation_project_id", "relation", ["project_id"])
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Add project_id to observation table
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
# Step 1: Add project_id column as nullable first
|
||||
op.add_column("observation", sa.Column("project_id", sa.Integer(), nullable=True))
|
||||
|
||||
# Step 2: Backfill project_id from entity.project_id via entity_id
|
||||
if dialect == "postgresql":
|
||||
op.execute("""
|
||||
UPDATE observation
|
||||
SET project_id = entity.project_id
|
||||
FROM entity
|
||||
WHERE observation.entity_id = entity.id
|
||||
""")
|
||||
else:
|
||||
# SQLite syntax
|
||||
op.execute("""
|
||||
UPDATE observation
|
||||
SET project_id = (
|
||||
SELECT entity.project_id
|
||||
FROM entity
|
||||
WHERE entity.id = observation.entity_id
|
||||
)
|
||||
""")
|
||||
|
||||
# Step 3: Make project_id NOT NULL and add foreign key
|
||||
if dialect == "postgresql":
|
||||
op.alter_column("observation", "project_id", nullable=False)
|
||||
op.create_foreign_key(
|
||||
"fk_observation_project_id",
|
||||
"observation",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
else:
|
||||
# SQLite requires batch operations for ALTER COLUMN
|
||||
with op.batch_alter_table("observation") as batch_op:
|
||||
batch_op.alter_column("project_id", nullable=False)
|
||||
batch_op.create_foreign_key(
|
||||
"fk_observation_project_id",
|
||||
"project",
|
||||
["project_id"],
|
||||
["id"],
|
||||
)
|
||||
|
||||
# Step 4: Create index on observation.project_id
|
||||
op.create_index("ix_observation_project_id", "observation", ["project_id"])
|
||||
|
||||
# Postgres-specific: pg_trgm and GIN indexes
|
||||
if dialect == "postgresql":
|
||||
# Enable pg_trgm extension for fuzzy string matching
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS pg_trgm")
|
||||
|
||||
# Create trigram indexes on entity table for fuzzy matching
|
||||
# GIN indexes with gin_trgm_ops support similarity searches
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_entity_title_trgm
|
||||
ON entity USING gin (title gin_trgm_ops)
|
||||
""")
|
||||
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_entity_permalink_trgm
|
||||
ON entity USING gin (permalink gin_trgm_ops)
|
||||
""")
|
||||
|
||||
# Create partial index on unresolved relations for efficient bulk resolution
|
||||
# This makes "WHERE to_id IS NULL AND project_id = X" queries very fast
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_relation_unresolved
|
||||
ON relation (project_id, to_name)
|
||||
WHERE to_id IS NULL
|
||||
""")
|
||||
|
||||
# Create index on relation.to_name for join performance in bulk resolution
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_relation_to_name
|
||||
ON relation (to_name)
|
||||
""")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Remove project_id from relation/observation and pg_trgm indexes."""
|
||||
connection = op.get_bind()
|
||||
dialect = connection.dialect.name
|
||||
|
||||
if dialect == "postgresql":
|
||||
# Drop Postgres-specific indexes
|
||||
op.execute("DROP INDEX IF EXISTS idx_relation_to_name")
|
||||
op.execute("DROP INDEX IF EXISTS idx_relation_unresolved")
|
||||
op.execute("DROP INDEX IF EXISTS idx_entity_permalink_trgm")
|
||||
op.execute("DROP INDEX IF EXISTS idx_entity_title_trgm")
|
||||
# Note: We don't drop the pg_trgm extension as other code may depend on it
|
||||
|
||||
# Drop project_id from observation
|
||||
op.drop_index("ix_observation_project_id", table_name="observation")
|
||||
op.drop_constraint("fk_observation_project_id", "observation", type_="foreignkey")
|
||||
op.drop_column("observation", "project_id")
|
||||
|
||||
# Drop project_id from relation
|
||||
op.drop_index("ix_relation_project_id", table_name="relation")
|
||||
op.drop_constraint("fk_relation_project_id", "relation", type_="foreignkey")
|
||||
op.drop_column("relation", "project_id")
|
||||
else:
|
||||
# SQLite requires batch operations
|
||||
op.drop_index("ix_observation_project_id", table_name="observation")
|
||||
with op.batch_alter_table("observation") as batch_op:
|
||||
batch_op.drop_constraint("fk_observation_project_id", type_="foreignkey")
|
||||
batch_op.drop_column("project_id")
|
||||
|
||||
op.drop_index("ix_relation_project_id", table_name="relation")
|
||||
with op.batch_alter_table("relation") as batch_op:
|
||||
batch_op.drop_constraint("fk_relation_project_id", type_="foreignkey")
|
||||
batch_op.drop_column("project_id")
|
||||
@@ -20,7 +20,17 @@ from basic_memory.api.routers import (
|
||||
search,
|
||||
prompt_router,
|
||||
)
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.api.v2.routers import (
|
||||
knowledge_router as v2_knowledge,
|
||||
project_router as v2_project,
|
||||
memory_router as v2_memory,
|
||||
search_router as v2_search,
|
||||
resource_router as v2_resource,
|
||||
directory_router as v2_directory,
|
||||
prompt_router as v2_prompt,
|
||||
importer_router as v2_importer,
|
||||
)
|
||||
from basic_memory.config import ConfigManager, init_api_logging
|
||||
from basic_memory.services.initialization import initialize_file_sync, initialize_app
|
||||
|
||||
|
||||
@@ -28,6 +38,9 @@ from basic_memory.services.initialization import initialize_file_sync, initializ
|
||||
async def lifespan(app: FastAPI): # pragma: no cover
|
||||
"""Lifecycle manager for the FastAPI app. Not called in stdio mcp mode"""
|
||||
|
||||
# Initialize logging for API (stdout in cloud mode, file otherwise)
|
||||
init_api_logging()
|
||||
|
||||
app_config = ConfigManager().config
|
||||
logger.info("Starting Basic Memory API")
|
||||
|
||||
@@ -46,6 +59,7 @@ async def lifespan(app: FastAPI): # pragma: no cover
|
||||
app.state.sync_task = asyncio.create_task(initialize_file_sync(app_config))
|
||||
else:
|
||||
logger.info("Sync changes disabled. Skipping file sync service.")
|
||||
app.state.sync_task = None
|
||||
|
||||
# proceed with startup
|
||||
yield
|
||||
@@ -54,6 +68,10 @@ async def lifespan(app: FastAPI): # pragma: no cover
|
||||
if app.state.sync_task:
|
||||
logger.info("Stopping sync...")
|
||||
app.state.sync_task.cancel() # pyright: ignore
|
||||
try:
|
||||
await app.state.sync_task
|
||||
except asyncio.CancelledError:
|
||||
logger.info("Sync task cancelled successfully")
|
||||
|
||||
await db.shutdown_db()
|
||||
|
||||
@@ -66,8 +84,7 @@ app = FastAPI(
|
||||
lifespan=lifespan,
|
||||
)
|
||||
|
||||
|
||||
# Include routers
|
||||
# Include v1 routers
|
||||
app.include_router(knowledge.router, prefix="/{project}")
|
||||
app.include_router(memory.router, prefix="/{project}")
|
||||
app.include_router(resource.router, prefix="/{project}")
|
||||
@@ -77,12 +94,20 @@ app.include_router(directory_router.router, prefix="/{project}")
|
||||
app.include_router(prompt_router.router, prefix="/{project}")
|
||||
app.include_router(importer_router.router, prefix="/{project}")
|
||||
|
||||
# Project resource router works accross projects
|
||||
# Include v2 routers (ID-based paths)
|
||||
app.include_router(v2_knowledge, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_memory, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_search, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_resource, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_directory, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_prompt, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_importer, prefix="/v2/projects/{project_id}")
|
||||
app.include_router(v2_project, prefix="/v2")
|
||||
|
||||
# Project resource router works across projects
|
||||
app.include_router(project.project_resource_router)
|
||||
app.include_router(management.router)
|
||||
|
||||
# Auth routes are handled by FastMCP automatically when auth is enabled
|
||||
|
||||
|
||||
@app.exception_handler(Exception)
|
||||
async def exception_handler(request, exc): # pragma: no cover
|
||||
|
||||
@@ -1,4 +1,11 @@
|
||||
"""Router for knowledge graph operations."""
|
||||
"""Router for knowledge graph operations.
|
||||
|
||||
⚠️ DEPRECATED: This v1 API is deprecated and will be removed on June 30, 2026.
|
||||
Please migrate to /v2/{project}/knowledge endpoints which use entity IDs instead
|
||||
of path-based identifiers for improved performance and stability.
|
||||
|
||||
Migration guide: See docs/migration/v1-to-v2.md
|
||||
"""
|
||||
|
||||
from typing import Annotated
|
||||
|
||||
@@ -25,7 +32,11 @@ from basic_memory.schemas import (
|
||||
from basic_memory.schemas.request import EditEntityRequest, MoveEntityRequest
|
||||
from basic_memory.schemas.base import Permalink, Entity
|
||||
|
||||
router = APIRouter(prefix="/knowledge", tags=["knowledge"])
|
||||
router = APIRouter(
|
||||
prefix="/knowledge",
|
||||
tags=["knowledge"],
|
||||
deprecated=True, # Marks entire router as deprecated in OpenAPI docs
|
||||
)
|
||||
|
||||
|
||||
async def resolve_relations_background(sync_service, entity_id: int, entity_permalink: str) -> None:
|
||||
|
||||
@@ -50,6 +50,7 @@ async def get_project(
|
||||
) # pragma: no cover
|
||||
|
||||
return ProjectItem(
|
||||
id=found_project.id,
|
||||
name=found_project.name,
|
||||
path=normalize_project_path(found_project.path),
|
||||
is_default=found_project.is_default or False,
|
||||
@@ -80,9 +81,17 @@ async def update_project(
|
||||
raise HTTPException(status_code=400, detail="Path must be absolute")
|
||||
|
||||
# Get original project info for the response
|
||||
old_project = await project_service.get_project(name)
|
||||
if not old_project:
|
||||
raise HTTPException(
|
||||
status_code=400, detail=f"Project '{name}' not found in configuration"
|
||||
)
|
||||
|
||||
old_project_info = ProjectItem(
|
||||
name=name,
|
||||
path=project_service.projects.get(name, ""),
|
||||
id=old_project.id,
|
||||
name=old_project.name,
|
||||
path=old_project.path,
|
||||
is_default=old_project.is_default or False,
|
||||
)
|
||||
|
||||
if path:
|
||||
@@ -91,14 +100,21 @@ async def update_project(
|
||||
await project_service.update_project(name, is_active=is_active)
|
||||
|
||||
# Get updated project info
|
||||
updated_path = path if path else project_service.projects.get(name, "")
|
||||
updated_project = await project_service.get_project(name)
|
||||
if not updated_project:
|
||||
raise HTTPException(status_code=404, detail=f"Project '{name}' not found after update")
|
||||
|
||||
return ProjectStatusResponse(
|
||||
message=f"Project '{name}' updated successfully",
|
||||
status="success",
|
||||
default=(name == project_service.default_project),
|
||||
old_project=old_project_info,
|
||||
new_project=ProjectItem(name=name, path=updated_path),
|
||||
new_project=ProjectItem(
|
||||
id=updated_project.id,
|
||||
name=updated_project.name,
|
||||
path=updated_project.path,
|
||||
is_default=updated_project.is_default or False,
|
||||
),
|
||||
)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
@@ -186,6 +202,7 @@ async def list_projects(
|
||||
|
||||
project_items = [
|
||||
ProjectItem(
|
||||
id=project.id,
|
||||
name=project.name,
|
||||
path=normalize_project_path(project.path),
|
||||
is_default=project.is_default or False,
|
||||
@@ -232,6 +249,7 @@ async def add_project(
|
||||
status="success",
|
||||
default=existing_project.is_default or False,
|
||||
new_project=ProjectItem(
|
||||
id=existing_project.id,
|
||||
name=existing_project.name,
|
||||
path=existing_project.path,
|
||||
is_default=existing_project.is_default or False,
|
||||
@@ -250,12 +268,20 @@ async def add_project(
|
||||
project_data.name, project_data.path, set_default=project_data.set_default
|
||||
)
|
||||
|
||||
# Fetch the newly created project to get its ID
|
||||
new_project = await project_service.get_project(project_data.name)
|
||||
if not new_project:
|
||||
raise HTTPException(status_code=500, detail="Failed to retrieve newly created project")
|
||||
|
||||
return ProjectStatusResponse( # pyright: ignore [reportCallIssue]
|
||||
message=f"Project '{project_data.name}' added successfully",
|
||||
status="success",
|
||||
default=project_data.set_default,
|
||||
new_project=ProjectItem(
|
||||
name=project_data.name, path=project_data.path, is_default=project_data.set_default
|
||||
id=new_project.id,
|
||||
name=new_project.name,
|
||||
path=new_project.path,
|
||||
is_default=new_project.is_default or False,
|
||||
),
|
||||
)
|
||||
except ValueError as e: # pragma: no cover
|
||||
@@ -306,7 +332,12 @@ async def remove_project(
|
||||
message=f"Project '{name}' removed successfully",
|
||||
status="success",
|
||||
default=False,
|
||||
old_project=ProjectItem(name=old_project.name, path=old_project.path),
|
||||
old_project=ProjectItem(
|
||||
id=old_project.id,
|
||||
name=old_project.name,
|
||||
path=old_project.path,
|
||||
is_default=old_project.is_default or False,
|
||||
),
|
||||
new_project=None,
|
||||
)
|
||||
except ValueError as e: # pragma: no cover
|
||||
@@ -349,8 +380,14 @@ async def set_default_project(
|
||||
message=f"Project '{name}' set as default successfully",
|
||||
status="success",
|
||||
default=True,
|
||||
old_project=ProjectItem(name=default_name, path=default_project.path),
|
||||
old_project=ProjectItem(
|
||||
id=default_project.id,
|
||||
name=default_name,
|
||||
path=default_project.path,
|
||||
is_default=False,
|
||||
),
|
||||
new_project=ProjectItem(
|
||||
id=new_default_project.id,
|
||||
name=name,
|
||||
path=new_default_project.path,
|
||||
is_default=True,
|
||||
@@ -378,7 +415,12 @@ async def get_default_project(
|
||||
status_code=404, detail=f"Default Project: '{default_name}' does not exist"
|
||||
)
|
||||
|
||||
return ProjectItem(name=default_project.name, path=default_project.path, is_default=True)
|
||||
return ProjectItem(
|
||||
id=default_project.id,
|
||||
name=default_project.name,
|
||||
path=default_project.path,
|
||||
is_default=True,
|
||||
)
|
||||
|
||||
|
||||
# Synchronize projects between config and database
|
||||
|
||||
@@ -2,9 +2,9 @@
|
||||
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Annotated
|
||||
from typing import Annotated, Union
|
||||
|
||||
from fastapi import APIRouter, HTTPException, BackgroundTasks, Body
|
||||
from fastapi import APIRouter, HTTPException, BackgroundTasks, Body, Response
|
||||
from fastapi.responses import FileResponse, JSONResponse
|
||||
from loguru import logger
|
||||
|
||||
@@ -25,6 +25,17 @@ from datetime import datetime
|
||||
router = APIRouter(prefix="/resource", tags=["resources"])
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: EntityModel) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
Returns the file's actual modification time, falling back to updated_at
|
||||
if mtime is not available.
|
||||
"""
|
||||
if entity.mtime:
|
||||
return datetime.fromtimestamp(entity.mtime).astimezone()
|
||||
return entity.updated_at
|
||||
|
||||
|
||||
def get_entity_ids(item: SearchIndexRow) -> set[int]:
|
||||
match item.type:
|
||||
case SearchItemType.ENTITY:
|
||||
@@ -39,7 +50,7 @@ def get_entity_ids(item: SearchIndexRow) -> set[int]:
|
||||
raise ValueError(f"Unexpected type: {item.type}")
|
||||
|
||||
|
||||
@router.get("/{identifier:path}")
|
||||
@router.get("/{identifier:path}", response_model=None)
|
||||
async def get_resource_content(
|
||||
config: ProjectConfigDep,
|
||||
link_resolver: LinkResolverDep,
|
||||
@@ -50,7 +61,7 @@ async def get_resource_content(
|
||||
identifier: str,
|
||||
page: int = 1,
|
||||
page_size: int = 10,
|
||||
) -> FileResponse:
|
||||
) -> Union[Response, FileResponse]:
|
||||
"""Get resource content by identifier: name or permalink."""
|
||||
logger.debug(f"Getting content for: {identifier}")
|
||||
|
||||
@@ -81,13 +92,16 @@ async def get_resource_content(
|
||||
# return single response
|
||||
if len(results) == 1:
|
||||
entity = results[0]
|
||||
file_path = Path(f"{config.home}/{entity.file_path}")
|
||||
if not file_path.exists():
|
||||
# Check file exists via file_service (for cloud compatibility)
|
||||
if not await file_service.exists(entity.file_path):
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"File not found: {file_path}",
|
||||
detail=f"File not found: {entity.file_path}",
|
||||
)
|
||||
return FileResponse(path=file_path)
|
||||
# Read content via file_service as bytes (works with both local and S3)
|
||||
content = await file_service.read_file_bytes(entity.file_path)
|
||||
content_type = file_service.content_type(entity.file_path)
|
||||
return Response(content=content, media_type=content_type)
|
||||
|
||||
# for multiple files, initialize a temporary file for writing the results
|
||||
with tempfile.NamedTemporaryFile(delete=False, mode="w", suffix=".md") as tmp_file:
|
||||
@@ -97,7 +111,7 @@ async def get_resource_content(
|
||||
# Read content for each entity
|
||||
content = await file_service.read_entity_content(result)
|
||||
memory_url = normalize_memory_url(result.permalink)
|
||||
modified_date = result.updated_at.isoformat()
|
||||
modified_date = _mtime_to_datetime(result).isoformat()
|
||||
checksum = result.checksum[:8] if result.checksum else ""
|
||||
|
||||
# Prepare the delimited content
|
||||
@@ -171,21 +185,17 @@ async def write_resource(
|
||||
else:
|
||||
content_str = str(content)
|
||||
|
||||
# Get full file path
|
||||
full_path = Path(f"{config.home}/{file_path}")
|
||||
|
||||
# Ensure parent directory exists
|
||||
full_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Write content to file
|
||||
checksum = await file_service.write_file(full_path, content_str)
|
||||
# Cloud compatibility: do not assume a local filesystem path structure.
|
||||
# Delegate directory creation + writes to the configured FileService (local or S3).
|
||||
await file_service.ensure_directory(Path(file_path).parent)
|
||||
checksum = await file_service.write_file(file_path, content_str)
|
||||
|
||||
# Get file info
|
||||
file_stats = file_service.file_stats(full_path)
|
||||
file_metadata = await file_service.get_file_metadata(file_path)
|
||||
|
||||
# Determine file details
|
||||
file_name = Path(file_path).name
|
||||
content_type = file_service.content_type(full_path)
|
||||
content_type = file_service.content_type(file_path)
|
||||
|
||||
entity_type = "canvas" if file_path.endswith(".canvas") else "file"
|
||||
|
||||
@@ -202,7 +212,7 @@ async def write_resource(
|
||||
"content_type": content_type,
|
||||
"file_path": file_path,
|
||||
"checksum": checksum,
|
||||
"updated_at": datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
"updated_at": file_metadata.modified_at,
|
||||
},
|
||||
)
|
||||
status_code = 200
|
||||
@@ -214,8 +224,8 @@ async def write_resource(
|
||||
content_type=content_type,
|
||||
file_path=file_path,
|
||||
checksum=checksum,
|
||||
created_at=datetime.fromtimestamp(file_stats.st_ctime).astimezone(),
|
||||
updated_at=datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
created_at=file_metadata.created_at,
|
||||
updated_at=file_metadata.modified_at,
|
||||
)
|
||||
entity = await entity_repository.add(entity)
|
||||
status_code = 201
|
||||
@@ -229,9 +239,9 @@ async def write_resource(
|
||||
content={
|
||||
"file_path": file_path,
|
||||
"checksum": checksum,
|
||||
"size": file_stats.st_size,
|
||||
"created_at": file_stats.st_ctime,
|
||||
"modified_at": file_stats.st_mtime,
|
||||
"size": file_metadata.size,
|
||||
"created_at": file_metadata.created_at.timestamp(),
|
||||
"modified_at": file_metadata.modified_at.timestamp(),
|
||||
},
|
||||
)
|
||||
except Exception as e: # pragma: no cover
|
||||
|
||||
@@ -24,11 +24,30 @@ async def to_graph_context(
|
||||
page: Optional[int] = None,
|
||||
page_size: Optional[int] = None,
|
||||
):
|
||||
# First pass: collect all entity IDs needed for relations
|
||||
entity_ids_needed: set[int] = set()
|
||||
for context_item in context_result.results:
|
||||
for item in (
|
||||
[context_item.primary_result] + context_item.observations + context_item.related_results
|
||||
):
|
||||
if item.type == SearchItemType.RELATION:
|
||||
if item.from_id: # pyright: ignore
|
||||
entity_ids_needed.add(item.from_id) # pyright: ignore
|
||||
if item.to_id:
|
||||
entity_ids_needed.add(item.to_id)
|
||||
|
||||
# Batch fetch all entities at once
|
||||
entity_lookup: dict[int, str] = {}
|
||||
if entity_ids_needed:
|
||||
entities = await entity_repository.find_by_ids(list(entity_ids_needed))
|
||||
entity_lookup = {e.id: e.title for e in entities}
|
||||
|
||||
# Helper function to convert items to summaries
|
||||
async def to_summary(item: SearchIndexRow | ContextResultRow):
|
||||
def to_summary(item: SearchIndexRow | ContextResultRow):
|
||||
match item.type:
|
||||
case SearchItemType.ENTITY:
|
||||
return EntitySummary(
|
||||
entity_id=item.id,
|
||||
title=item.title, # pyright: ignore
|
||||
permalink=item.permalink,
|
||||
content=item.content,
|
||||
@@ -37,6 +56,8 @@ async def to_graph_context(
|
||||
)
|
||||
case SearchItemType.OBSERVATION:
|
||||
return ObservationSummary(
|
||||
observation_id=item.id,
|
||||
entity_id=item.entity_id, # pyright: ignore
|
||||
title=item.title, # pyright: ignore
|
||||
file_path=item.file_path,
|
||||
category=item.category, # pyright: ignore
|
||||
@@ -45,15 +66,19 @@ async def to_graph_context(
|
||||
created_at=item.created_at,
|
||||
)
|
||||
case SearchItemType.RELATION:
|
||||
from_entity = await entity_repository.find_by_id(item.from_id) # pyright: ignore
|
||||
to_entity = await entity_repository.find_by_id(item.to_id) if item.to_id else None
|
||||
from_title = entity_lookup.get(item.from_id) if item.from_id else None # pyright: ignore
|
||||
to_title = entity_lookup.get(item.to_id) if item.to_id else None
|
||||
return RelationSummary(
|
||||
relation_id=item.id,
|
||||
entity_id=item.entity_id, # pyright: ignore
|
||||
title=item.title, # pyright: ignore
|
||||
file_path=item.file_path,
|
||||
permalink=item.permalink, # pyright: ignore
|
||||
relation_type=item.relation_type, # pyright: ignore
|
||||
from_entity=from_entity.title if from_entity else None,
|
||||
to_entity=to_entity.title if to_entity else None,
|
||||
from_entity=from_title,
|
||||
from_entity_id=item.from_id, # pyright: ignore
|
||||
to_entity=to_title,
|
||||
to_entity_id=item.to_id,
|
||||
created_at=item.created_at,
|
||||
)
|
||||
case _: # pragma: no cover
|
||||
@@ -63,23 +88,19 @@ async def to_graph_context(
|
||||
hierarchical_results = []
|
||||
for context_item in context_result.results:
|
||||
# Process primary result
|
||||
primary_result = await to_summary(context_item.primary_result)
|
||||
primary_result = to_summary(context_item.primary_result)
|
||||
|
||||
# Process observations
|
||||
observations = []
|
||||
for obs in context_item.observations:
|
||||
observations.append(await to_summary(obs))
|
||||
# Process observations (always ObservationSummary, validated by context_service)
|
||||
observations = [to_summary(obs) for obs in context_item.observations]
|
||||
|
||||
# Process related results
|
||||
related = []
|
||||
for rel in context_item.related_results:
|
||||
related.append(await to_summary(rel))
|
||||
related = [to_summary(rel) for rel in context_item.related_results]
|
||||
|
||||
# Add to hierarchical results
|
||||
hierarchical_results.append(
|
||||
ContextResult(
|
||||
primary_result=primary_result,
|
||||
observations=observations,
|
||||
observations=observations, # pyright: ignore[reportArgumentType]
|
||||
related_results=related,
|
||||
)
|
||||
)
|
||||
@@ -111,6 +132,21 @@ async def to_search_results(entity_service: EntityService, results: List[SearchI
|
||||
search_results = []
|
||||
for r in results:
|
||||
entities = await entity_service.get_entities_by_id([r.entity_id, r.from_id, r.to_id]) # pyright: ignore
|
||||
|
||||
# Determine which IDs to set based on type
|
||||
entity_id = None
|
||||
observation_id = None
|
||||
relation_id = None
|
||||
|
||||
if r.type == SearchItemType.ENTITY:
|
||||
entity_id = r.id
|
||||
elif r.type == SearchItemType.OBSERVATION:
|
||||
observation_id = r.id
|
||||
entity_id = r.entity_id # Parent entity
|
||||
elif r.type == SearchItemType.RELATION:
|
||||
relation_id = r.id
|
||||
entity_id = r.entity_id # Parent entity
|
||||
|
||||
search_results.append(
|
||||
SearchResult(
|
||||
title=r.title, # pyright: ignore
|
||||
@@ -121,6 +157,9 @@ async def to_search_results(entity_service: EntityService, results: List[SearchI
|
||||
content=r.content,
|
||||
file_path=r.file_path,
|
||||
metadata=r.metadata,
|
||||
entity_id=entity_id,
|
||||
observation_id=observation_id,
|
||||
relation_id=relation_id,
|
||||
category=r.category,
|
||||
from_entity=entities[0].permalink if entities else None,
|
||||
to_entity=entities[1].permalink if len(entities) > 1 else None,
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
"""API v2 module - ID-based entity references.
|
||||
|
||||
Version 2 of the Basic Memory API uses integer entity IDs as the primary
|
||||
identifier for improved performance and stability.
|
||||
|
||||
Key changes from v1:
|
||||
- Entity lookups use integer IDs instead of paths/permalinks
|
||||
- Direct database queries instead of cascading resolution
|
||||
- Stable references that don't change with file moves
|
||||
- Better caching support
|
||||
|
||||
All v2 routers are registered with the /v2 prefix.
|
||||
"""
|
||||
|
||||
from basic_memory.api.v2.routers import (
|
||||
knowledge_router,
|
||||
memory_router,
|
||||
project_router,
|
||||
resource_router,
|
||||
search_router,
|
||||
directory_router,
|
||||
prompt_router,
|
||||
importer_router,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"knowledge_router",
|
||||
"memory_router",
|
||||
"project_router",
|
||||
"resource_router",
|
||||
"search_router",
|
||||
"directory_router",
|
||||
"prompt_router",
|
||||
"importer_router",
|
||||
]
|
||||
@@ -0,0 +1,21 @@
|
||||
"""V2 API routers."""
|
||||
|
||||
from basic_memory.api.v2.routers.knowledge_router import router as knowledge_router
|
||||
from basic_memory.api.v2.routers.project_router import router as project_router
|
||||
from basic_memory.api.v2.routers.memory_router import router as memory_router
|
||||
from basic_memory.api.v2.routers.search_router import router as search_router
|
||||
from basic_memory.api.v2.routers.resource_router import router as resource_router
|
||||
from basic_memory.api.v2.routers.directory_router import router as directory_router
|
||||
from basic_memory.api.v2.routers.prompt_router import router as prompt_router
|
||||
from basic_memory.api.v2.routers.importer_router import router as importer_router
|
||||
|
||||
__all__ = [
|
||||
"knowledge_router",
|
||||
"project_router",
|
||||
"memory_router",
|
||||
"search_router",
|
||||
"resource_router",
|
||||
"directory_router",
|
||||
"prompt_router",
|
||||
"importer_router",
|
||||
]
|
||||
@@ -0,0 +1,93 @@
|
||||
"""V2 Directory Router - ID-based directory tree operations.
|
||||
|
||||
This router provides directory structure browsing for projects using
|
||||
integer project IDs instead of name-based identifiers.
|
||||
|
||||
Key improvements:
|
||||
- Direct project lookup via integer primary keys
|
||||
- Consistent with other v2 endpoints
|
||||
- Better performance through indexed queries
|
||||
"""
|
||||
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import APIRouter, Query
|
||||
|
||||
from basic_memory.deps import DirectoryServiceV2Dep, ProjectIdPathDep
|
||||
from basic_memory.schemas.directory import DirectoryNode
|
||||
|
||||
router = APIRouter(prefix="/directory", tags=["directory-v2"])
|
||||
|
||||
|
||||
@router.get("/tree", response_model=DirectoryNode, response_model_exclude_none=True)
|
||||
async def get_directory_tree(
|
||||
directory_service: DirectoryServiceV2Dep,
|
||||
project_id: ProjectIdPathDep,
|
||||
):
|
||||
"""Get hierarchical directory structure from the knowledge base.
|
||||
|
||||
Args:
|
||||
directory_service: Service for directory operations
|
||||
project_id: Numeric project ID
|
||||
|
||||
Returns:
|
||||
DirectoryNode representing the root of the hierarchical tree structure
|
||||
"""
|
||||
# Get a hierarchical directory tree for the specific project
|
||||
tree = await directory_service.get_directory_tree()
|
||||
|
||||
# Return the hierarchical tree
|
||||
return tree
|
||||
|
||||
|
||||
@router.get("/structure", response_model=DirectoryNode, response_model_exclude_none=True)
|
||||
async def get_directory_structure(
|
||||
directory_service: DirectoryServiceV2Dep,
|
||||
project_id: ProjectIdPathDep,
|
||||
):
|
||||
"""Get folder structure for navigation (no files).
|
||||
|
||||
Optimized endpoint for folder tree navigation. Returns only directory nodes
|
||||
without file metadata. For full tree with files, use /directory/tree.
|
||||
|
||||
Args:
|
||||
directory_service: Service for directory operations
|
||||
project_id: Numeric project ID
|
||||
|
||||
Returns:
|
||||
DirectoryNode tree containing only folders (type="directory")
|
||||
"""
|
||||
structure = await directory_service.get_directory_structure()
|
||||
return structure
|
||||
|
||||
|
||||
@router.get("/list", response_model=List[DirectoryNode], response_model_exclude_none=True)
|
||||
async def list_directory(
|
||||
directory_service: DirectoryServiceV2Dep,
|
||||
project_id: ProjectIdPathDep,
|
||||
dir_name: str = Query("/", description="Directory path to list"),
|
||||
depth: int = Query(1, ge=1, le=10, description="Recursion depth (1-10)"),
|
||||
file_name_glob: Optional[str] = Query(
|
||||
None, description="Glob pattern for filtering file names"
|
||||
),
|
||||
):
|
||||
"""List directory contents with filtering and depth control.
|
||||
|
||||
Args:
|
||||
directory_service: Service for directory operations
|
||||
project_id: Numeric project ID
|
||||
dir_name: Directory path to list (default: root "/")
|
||||
depth: Recursion depth (1-10, default: 1 for immediate children only)
|
||||
file_name_glob: Optional glob pattern for filtering file names (e.g., "*.md", "*meeting*")
|
||||
|
||||
Returns:
|
||||
List of DirectoryNode objects matching the criteria
|
||||
"""
|
||||
# Get directory listing with filtering
|
||||
nodes = await directory_service.list_directory(
|
||||
dir_name=dir_name,
|
||||
depth=depth,
|
||||
file_name_glob=file_name_glob,
|
||||
)
|
||||
|
||||
return nodes
|
||||
@@ -0,0 +1,182 @@
|
||||
"""V2 Import Router - ID-based data import operations.
|
||||
|
||||
This router uses v2 dependencies for consistent project ID handling.
|
||||
Import endpoints use project_id in the path for consistency with other v2 endpoints.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Form, HTTPException, UploadFile, status
|
||||
|
||||
from basic_memory.deps import (
|
||||
ChatGPTImporterV2Dep,
|
||||
ClaudeConversationsImporterV2Dep,
|
||||
ClaudeProjectsImporterV2Dep,
|
||||
MemoryJsonImporterV2Dep,
|
||||
ProjectIdPathDep,
|
||||
)
|
||||
from basic_memory.importers import Importer
|
||||
from basic_memory.schemas.importer import (
|
||||
ChatImportResult,
|
||||
EntityImportResult,
|
||||
ProjectImportResult,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/import", tags=["import-v2"])
|
||||
|
||||
|
||||
@router.post("/chatgpt", response_model=ChatImportResult)
|
||||
async def import_chatgpt(
|
||||
project_id: ProjectIdPathDep,
|
||||
importer: ChatGPTImporterV2Dep,
|
||||
file: UploadFile,
|
||||
folder: str = Form("conversations"),
|
||||
) -> ChatImportResult:
|
||||
"""Import conversations from ChatGPT JSON export.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
file: The ChatGPT conversations.json file.
|
||||
folder: The folder to place the files in.
|
||||
importer: ChatGPT importer instance.
|
||||
|
||||
Returns:
|
||||
ChatImportResult with import statistics.
|
||||
|
||||
Raises:
|
||||
HTTPException: If import fails.
|
||||
"""
|
||||
logger.info(f"V2 Importing ChatGPT conversations for project {project_id}")
|
||||
return await import_file(importer, file, folder)
|
||||
|
||||
|
||||
@router.post("/claude/conversations", response_model=ChatImportResult)
|
||||
async def import_claude_conversations(
|
||||
project_id: ProjectIdPathDep,
|
||||
importer: ClaudeConversationsImporterV2Dep,
|
||||
file: UploadFile,
|
||||
folder: str = Form("conversations"),
|
||||
) -> ChatImportResult:
|
||||
"""Import conversations from Claude conversations.json export.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
file: The Claude conversations.json file.
|
||||
folder: The folder to place the files in.
|
||||
importer: Claude conversations importer instance.
|
||||
|
||||
Returns:
|
||||
ChatImportResult with import statistics.
|
||||
|
||||
Raises:
|
||||
HTTPException: If import fails.
|
||||
"""
|
||||
logger.info(f"V2 Importing Claude conversations for project {project_id}")
|
||||
return await import_file(importer, file, folder)
|
||||
|
||||
|
||||
@router.post("/claude/projects", response_model=ProjectImportResult)
|
||||
async def import_claude_projects(
|
||||
project_id: ProjectIdPathDep,
|
||||
importer: ClaudeProjectsImporterV2Dep,
|
||||
file: UploadFile,
|
||||
folder: str = Form("projects"),
|
||||
) -> ProjectImportResult:
|
||||
"""Import projects from Claude projects.json export.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
file: The Claude projects.json file.
|
||||
folder: The base folder to place the files in.
|
||||
importer: Claude projects importer instance.
|
||||
|
||||
Returns:
|
||||
ProjectImportResult with import statistics.
|
||||
|
||||
Raises:
|
||||
HTTPException: If import fails.
|
||||
"""
|
||||
logger.info(f"V2 Importing Claude projects for project {project_id}")
|
||||
return await import_file(importer, file, folder)
|
||||
|
||||
|
||||
@router.post("/memory-json", response_model=EntityImportResult)
|
||||
async def import_memory_json(
|
||||
project_id: ProjectIdPathDep,
|
||||
importer: MemoryJsonImporterV2Dep,
|
||||
file: UploadFile,
|
||||
folder: str = Form("conversations"),
|
||||
) -> EntityImportResult:
|
||||
"""Import entities and relations from a memory.json file.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
file: The memory.json file.
|
||||
folder: Optional destination folder within the project.
|
||||
importer: Memory JSON importer instance.
|
||||
|
||||
Returns:
|
||||
EntityImportResult with import statistics.
|
||||
|
||||
Raises:
|
||||
HTTPException: If import fails.
|
||||
"""
|
||||
logger.info(f"V2 Importing memory.json for project {project_id}")
|
||||
try:
|
||||
file_data = []
|
||||
file_bytes = await file.read()
|
||||
file_str = file_bytes.decode("utf-8")
|
||||
for line in file_str.splitlines():
|
||||
json_data = json.loads(line)
|
||||
file_data.append(json_data)
|
||||
|
||||
result = await importer.import_data(file_data, folder)
|
||||
if not result.success: # pragma: no cover
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=result.error_message or "Import failed",
|
||||
)
|
||||
except Exception as e:
|
||||
logger.exception("V2 Import failed")
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=f"Import failed: {str(e)}",
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
async def import_file(importer: Importer, file: UploadFile, destination_folder: str):
|
||||
"""Helper function to import a file using an importer instance.
|
||||
|
||||
Args:
|
||||
importer: The importer instance to use
|
||||
file: The file to import
|
||||
destination_folder: Destination folder for imported content
|
||||
|
||||
Returns:
|
||||
Import result from the importer
|
||||
|
||||
Raises:
|
||||
HTTPException: If import fails
|
||||
"""
|
||||
try:
|
||||
# Process file
|
||||
json_data = json.load(file.file)
|
||||
result = await importer.import_data(json_data, destination_folder)
|
||||
if not result.success: # pragma: no cover
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=result.error_message or "Import failed",
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
logger.exception("V2 Import failed")
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=f"Import failed: {str(e)}",
|
||||
)
|
||||
@@ -0,0 +1,415 @@
|
||||
"""V2 Knowledge Router - ID-based entity operations.
|
||||
|
||||
This router provides ID-based CRUD operations for entities, replacing the
|
||||
path-based identifiers used in v1 with direct integer ID lookups.
|
||||
|
||||
Key improvements:
|
||||
- Direct database lookups via integer primary keys
|
||||
- Stable references that don't change with file moves
|
||||
- Better performance through indexed queries
|
||||
- Simplified caching strategies
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, BackgroundTasks, Depends, Response
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.deps import (
|
||||
EntityServiceV2Dep,
|
||||
SearchServiceV2Dep,
|
||||
LinkResolverV2Dep,
|
||||
ProjectConfigV2Dep,
|
||||
AppConfigDep,
|
||||
SyncServiceV2Dep,
|
||||
EntityRepositoryV2Dep,
|
||||
ProjectIdPathDep,
|
||||
)
|
||||
from basic_memory.schemas import DeleteEntitiesResponse
|
||||
from basic_memory.schemas.base import Entity
|
||||
from basic_memory.schemas.request import EditEntityRequest
|
||||
from basic_memory.schemas.v2 import (
|
||||
EntityResolveRequest,
|
||||
EntityResolveResponse,
|
||||
EntityResponseV2,
|
||||
MoveEntityRequestV2,
|
||||
)
|
||||
|
||||
router = APIRouter(prefix="/knowledge", tags=["knowledge-v2"])
|
||||
|
||||
|
||||
async def resolve_relations_background(sync_service, entity_id: int, entity_permalink: str) -> None:
|
||||
"""Background task to resolve relations for a specific entity.
|
||||
|
||||
This runs asynchronously after the API response is sent, preventing
|
||||
long delays when creating entities with many relations.
|
||||
"""
|
||||
try:
|
||||
# Only resolve relations for the newly created entity
|
||||
await sync_service.resolve_relations(entity_id=entity_id)
|
||||
logger.debug(
|
||||
f"Background: Resolved relations for entity {entity_permalink} (id={entity_id})"
|
||||
)
|
||||
except Exception as e:
|
||||
# Log but don't fail - this is a background task
|
||||
logger.warning(
|
||||
f"Background: Failed to resolve relations for entity {entity_permalink}: {e}"
|
||||
)
|
||||
|
||||
|
||||
## Resolution endpoint
|
||||
|
||||
|
||||
@router.post("/resolve", response_model=EntityResolveResponse)
|
||||
async def resolve_identifier(
|
||||
project_id: ProjectIdPathDep,
|
||||
data: EntityResolveRequest,
|
||||
link_resolver: LinkResolverV2Dep,
|
||||
) -> EntityResolveResponse:
|
||||
"""Resolve a string identifier (permalink, title, or path) to an entity ID.
|
||||
|
||||
This endpoint provides a bridge between v1-style identifiers and v2 entity IDs.
|
||||
Use this to convert existing references to the new ID-based format.
|
||||
|
||||
Args:
|
||||
data: Request containing the identifier to resolve
|
||||
|
||||
Returns:
|
||||
Entity ID and metadata about how it was resolved
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if identifier cannot be resolved
|
||||
|
||||
Example:
|
||||
POST /v2/{project}/knowledge/resolve
|
||||
{"identifier": "specs/search"}
|
||||
|
||||
Returns:
|
||||
{
|
||||
"entity_id": 123,
|
||||
"permalink": "specs/search",
|
||||
"file_path": "specs/search.md",
|
||||
"title": "Search Specification",
|
||||
"resolution_method": "permalink"
|
||||
}
|
||||
"""
|
||||
logger.info(f"API v2 request: resolve_identifier for '{data.identifier}'")
|
||||
|
||||
# Try to resolve the identifier
|
||||
entity = await link_resolver.resolve_link(data.identifier)
|
||||
if not entity:
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"Could not resolve identifier: '{data.identifier}'"
|
||||
)
|
||||
|
||||
# Determine resolution method
|
||||
resolution_method = "search" # default
|
||||
if data.identifier.isdigit():
|
||||
resolution_method = "id"
|
||||
elif entity.permalink == data.identifier:
|
||||
resolution_method = "permalink"
|
||||
elif entity.title == data.identifier:
|
||||
resolution_method = "title"
|
||||
elif entity.file_path == data.identifier:
|
||||
resolution_method = "path"
|
||||
|
||||
result = EntityResolveResponse(
|
||||
entity_id=entity.id,
|
||||
permalink=entity.permalink,
|
||||
file_path=entity.file_path,
|
||||
title=entity.title,
|
||||
resolution_method=resolution_method,
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"API v2 response: resolved '{data.identifier}' to entity_id={result.entity_id} via {resolution_method}"
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
## Read endpoints
|
||||
|
||||
|
||||
@router.get("/entities/{entity_id}", response_model=EntityResponseV2)
|
||||
async def get_entity_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
) -> EntityResponseV2:
|
||||
"""Get an entity by its numeric ID.
|
||||
|
||||
This is the primary entity retrieval method in v2, using direct database
|
||||
lookups for maximum performance.
|
||||
|
||||
Args:
|
||||
entity_id: Numeric entity ID
|
||||
|
||||
Returns:
|
||||
Complete entity with observations and relations
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if entity not found
|
||||
"""
|
||||
logger.info(f"API v2 request: get_entity_by_id entity_id={entity_id}")
|
||||
|
||||
entity = await entity_repository.get_by_id(entity_id)
|
||||
if not entity:
|
||||
raise HTTPException(status_code=404, detail=f"Entity {entity_id} not found")
|
||||
|
||||
result = EntityResponseV2.model_validate(entity)
|
||||
logger.info(f"API v2 response: entity_id={entity_id}, title='{result.title}'")
|
||||
|
||||
return result
|
||||
|
||||
|
||||
## Create endpoints
|
||||
|
||||
|
||||
@router.post("/entities", response_model=EntityResponseV2)
|
||||
async def create_entity(
|
||||
project_id: ProjectIdPathDep,
|
||||
data: Entity,
|
||||
background_tasks: BackgroundTasks,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
) -> EntityResponseV2:
|
||||
"""Create a new entity.
|
||||
|
||||
Args:
|
||||
data: Entity data to create
|
||||
|
||||
Returns:
|
||||
Created entity with generated ID
|
||||
"""
|
||||
logger.info(
|
||||
"API v2 request", endpoint="create_entity", entity_type=data.entity_type, title=data.title
|
||||
)
|
||||
|
||||
entity = await entity_service.create_entity(data)
|
||||
|
||||
# reindex
|
||||
await search_service.index_entity(entity, background_tasks=background_tasks)
|
||||
result = EntityResponseV2.model_validate(entity)
|
||||
|
||||
logger.info(
|
||||
f"API v2 response: endpoint='create_entity' id={entity.id}, title={result.title}, permalink={result.permalink}, status_code=201"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
## Update endpoints
|
||||
|
||||
|
||||
@router.put("/entities/{entity_id}", response_model=EntityResponseV2)
|
||||
async def update_entity_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
data: Entity,
|
||||
response: Response,
|
||||
background_tasks: BackgroundTasks,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
sync_service: SyncServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
) -> EntityResponseV2:
|
||||
"""Update an entity by ID.
|
||||
|
||||
If the entity doesn't exist, it will be created (upsert behavior).
|
||||
|
||||
Args:
|
||||
entity_id: Numeric entity ID
|
||||
data: Updated entity data
|
||||
|
||||
Returns:
|
||||
Updated entity
|
||||
"""
|
||||
logger.info(f"API v2 request: update_entity_by_id entity_id={entity_id}")
|
||||
|
||||
# Check if entity exists
|
||||
existing = await entity_repository.get_by_id(entity_id)
|
||||
created = existing is None
|
||||
|
||||
# Perform update or create
|
||||
entity, _ = await entity_service.create_or_update_entity(data)
|
||||
response.status_code = 201 if created else 200
|
||||
|
||||
# reindex
|
||||
await search_service.index_entity(entity, background_tasks=background_tasks)
|
||||
|
||||
# Schedule relation resolution for new entities
|
||||
if created:
|
||||
background_tasks.add_task(
|
||||
resolve_relations_background, sync_service, entity.id, entity.permalink or ""
|
||||
)
|
||||
|
||||
result = EntityResponseV2.model_validate(entity)
|
||||
|
||||
logger.info(
|
||||
f"API v2 response: entity_id={entity_id}, created={created}, status_code={response.status_code}"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
@router.patch("/entities/{entity_id}", response_model=EntityResponseV2)
|
||||
async def edit_entity_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
data: EditEntityRequest,
|
||||
background_tasks: BackgroundTasks,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
) -> EntityResponseV2:
|
||||
"""Edit an existing entity by ID using operations like append, prepend, etc.
|
||||
|
||||
Args:
|
||||
entity_id: Numeric entity ID
|
||||
data: Edit operation details
|
||||
|
||||
Returns:
|
||||
Updated entity
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if entity not found, 400 if edit fails
|
||||
"""
|
||||
logger.info(
|
||||
f"API v2 request: edit_entity_by_id entity_id={entity_id}, operation='{data.operation}'"
|
||||
)
|
||||
|
||||
# Verify entity exists
|
||||
entity = await entity_repository.get_by_id(entity_id)
|
||||
if not entity:
|
||||
raise HTTPException(status_code=404, detail=f"Entity {entity_id} not found")
|
||||
|
||||
try:
|
||||
# Edit using the entity's permalink or path
|
||||
identifier = entity.permalink or entity.file_path
|
||||
updated_entity = await entity_service.edit_entity(
|
||||
identifier=identifier,
|
||||
operation=data.operation,
|
||||
content=data.content,
|
||||
section=data.section,
|
||||
find_text=data.find_text,
|
||||
expected_replacements=data.expected_replacements,
|
||||
)
|
||||
|
||||
# Reindex
|
||||
await search_service.index_entity(updated_entity, background_tasks=background_tasks)
|
||||
|
||||
result = EntityResponseV2.model_validate(updated_entity)
|
||||
|
||||
logger.info(
|
||||
f"API v2 response: entity_id={entity_id}, operation='{data.operation}', status_code=200"
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error editing entity {entity_id}: {e}")
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
|
||||
|
||||
## Delete endpoints
|
||||
|
||||
|
||||
@router.delete("/entities/{entity_id}", response_model=DeleteEntitiesResponse)
|
||||
async def delete_entity_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
background_tasks: BackgroundTasks,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
search_service=Depends(lambda: None), # Optional for now
|
||||
) -> DeleteEntitiesResponse:
|
||||
"""Delete an entity by ID.
|
||||
|
||||
Args:
|
||||
entity_id: Numeric entity ID
|
||||
|
||||
Returns:
|
||||
Deletion status
|
||||
|
||||
Note: Returns deleted=False if entity doesn't exist (idempotent)
|
||||
"""
|
||||
logger.info(f"API v2 request: delete_entity_by_id entity_id={entity_id}")
|
||||
|
||||
entity = await entity_repository.get_by_id(entity_id)
|
||||
if entity is None:
|
||||
logger.info(f"API v2 response: entity_id={entity_id} not found, deleted=False")
|
||||
return DeleteEntitiesResponse(deleted=False)
|
||||
|
||||
# Delete the entity
|
||||
deleted = await entity_service.delete_entity(entity_id)
|
||||
|
||||
# Remove from search index if search service available
|
||||
if search_service:
|
||||
background_tasks.add_task(search_service.handle_delete, entity)
|
||||
|
||||
logger.info(f"API v2 response: entity_id={entity_id}, deleted={deleted}")
|
||||
|
||||
return DeleteEntitiesResponse(deleted=deleted)
|
||||
|
||||
|
||||
## Move endpoint
|
||||
|
||||
|
||||
@router.put("/entities/{entity_id}/move", response_model=EntityResponseV2)
|
||||
async def move_entity(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
data: MoveEntityRequestV2,
|
||||
background_tasks: BackgroundTasks,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
project_config: ProjectConfigV2Dep,
|
||||
app_config: AppConfigDep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
) -> EntityResponseV2:
|
||||
"""Move an entity to a new file location.
|
||||
|
||||
V2 API uses entity ID in the URL path for stable references.
|
||||
The entity ID will remain stable after the move.
|
||||
|
||||
Args:
|
||||
project_id: Project ID from URL path
|
||||
entity_id: Entity ID from URL path (primary identifier)
|
||||
data: Move request with destination path only
|
||||
|
||||
Returns:
|
||||
Updated entity with new file path
|
||||
"""
|
||||
logger.info(
|
||||
f"API v2 request: move_entity entity_id={entity_id}, destination='{data.destination_path}'"
|
||||
)
|
||||
|
||||
try:
|
||||
# First, get the entity by ID to verify it exists
|
||||
entity = await entity_repository.find_by_id(entity_id)
|
||||
if not entity:
|
||||
raise HTTPException(status_code=404, detail=f"Entity not found: {entity_id}")
|
||||
|
||||
# Move the entity using its current file path as identifier
|
||||
moved_entity = await entity_service.move_entity(
|
||||
identifier=entity.file_path, # Use file path for resolution
|
||||
destination_path=data.destination_path,
|
||||
project_config=project_config,
|
||||
app_config=app_config,
|
||||
)
|
||||
|
||||
# Reindex at new location
|
||||
reindexed_entity = await entity_service.link_resolver.resolve_link(data.destination_path)
|
||||
if reindexed_entity:
|
||||
await search_service.index_entity(reindexed_entity, background_tasks=background_tasks)
|
||||
|
||||
result = EntityResponseV2.model_validate(moved_entity)
|
||||
|
||||
logger.info(
|
||||
f"API v2 response: moved entity_id={moved_entity.id} to '{data.destination_path}'"
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Error moving entity: {e}")
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
@@ -0,0 +1,130 @@
|
||||
"""V2 routes for memory:// URI operations.
|
||||
|
||||
This router uses integer project IDs for stable, efficient routing.
|
||||
V1 uses string-based project names which are less efficient and less stable.
|
||||
"""
|
||||
|
||||
from typing import Annotated, Optional
|
||||
|
||||
from fastapi import APIRouter, Query
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.deps import ContextServiceV2Dep, EntityRepositoryV2Dep, ProjectIdPathDep
|
||||
from basic_memory.schemas.base import TimeFrame, parse_timeframe
|
||||
from basic_memory.schemas.memory import (
|
||||
GraphContext,
|
||||
normalize_memory_url,
|
||||
)
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
from basic_memory.api.routers.utils import to_graph_context
|
||||
|
||||
# Note: No prefix here - it's added during registration as /v2/{project_id}/memory
|
||||
router = APIRouter(tags=["memory"])
|
||||
|
||||
|
||||
@router.get("/memory/recent", response_model=GraphContext)
|
||||
async def recent(
|
||||
project_id: ProjectIdPathDep,
|
||||
context_service: ContextServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
type: Annotated[list[SearchItemType] | None, Query()] = None,
|
||||
depth: int = 1,
|
||||
timeframe: TimeFrame = "7d",
|
||||
page: int = 1,
|
||||
page_size: int = 10,
|
||||
max_related: int = 10,
|
||||
) -> GraphContext:
|
||||
"""Get recent activity context for a project.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
context_service: Context service scoped to project
|
||||
entity_repository: Entity repository scoped to project
|
||||
type: Types of items to include (entities, relations, observations)
|
||||
depth: How many levels of related entities to include
|
||||
timeframe: Time window for recent activity (e.g., "7d", "1 week")
|
||||
page: Page number for pagination
|
||||
page_size: Number of items per page
|
||||
max_related: Maximum related entities to include per item
|
||||
|
||||
Returns:
|
||||
GraphContext with recent activity and related entities
|
||||
"""
|
||||
# return all types by default
|
||||
types = (
|
||||
[SearchItemType.ENTITY, SearchItemType.RELATION, SearchItemType.OBSERVATION]
|
||||
if not type
|
||||
else type
|
||||
)
|
||||
|
||||
logger.debug(
|
||||
f"V2 Getting recent context for project {project_id}: `{types}` depth: `{depth}` timeframe: `{timeframe}` page: `{page}` page_size: `{page_size}` max_related: `{max_related}`"
|
||||
)
|
||||
# Parse timeframe
|
||||
since = parse_timeframe(timeframe)
|
||||
limit = page_size
|
||||
offset = (page - 1) * page_size
|
||||
|
||||
# Build context
|
||||
context = await context_service.build_context(
|
||||
types=types, depth=depth, since=since, limit=limit, offset=offset, max_related=max_related
|
||||
)
|
||||
recent_context = await to_graph_context(
|
||||
context, entity_repository=entity_repository, page=page, page_size=page_size
|
||||
)
|
||||
logger.debug(f"V2 Recent context: {recent_context.model_dump_json()}")
|
||||
return recent_context
|
||||
|
||||
|
||||
# get_memory_context needs to be declared last so other paths can match
|
||||
|
||||
|
||||
@router.get("/memory/{uri:path}", response_model=GraphContext)
|
||||
async def get_memory_context(
|
||||
project_id: ProjectIdPathDep,
|
||||
context_service: ContextServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
uri: str,
|
||||
depth: int = 1,
|
||||
timeframe: Optional[TimeFrame] = None,
|
||||
page: int = 1,
|
||||
page_size: int = 10,
|
||||
max_related: int = 10,
|
||||
) -> GraphContext:
|
||||
"""Get rich context from memory:// URI.
|
||||
|
||||
V2 supports both legacy path-based URIs and new ID-based URIs:
|
||||
- Legacy: memory://path/to/note
|
||||
- ID-based: memory://id/123 or memory://123
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
context_service: Context service scoped to project
|
||||
entity_repository: Entity repository scoped to project
|
||||
uri: Memory URI path (e.g., "id/123", "123", or "path/to/note")
|
||||
depth: How many levels of related entities to include
|
||||
timeframe: Optional time window for filtering related content
|
||||
page: Page number for pagination
|
||||
page_size: Number of items per page
|
||||
max_related: Maximum related entities to include
|
||||
|
||||
Returns:
|
||||
GraphContext with the entity and its related context
|
||||
"""
|
||||
logger.debug(
|
||||
f"V2 Getting context for project {project_id}, URI: `{uri}` depth: `{depth}` timeframe: `{timeframe}` page: `{page}` page_size: `{page_size}` max_related: `{max_related}`"
|
||||
)
|
||||
memory_url = normalize_memory_url(uri)
|
||||
|
||||
# Parse timeframe
|
||||
since = parse_timeframe(timeframe) if timeframe else None
|
||||
limit = page_size
|
||||
offset = (page - 1) * page_size
|
||||
|
||||
# Build context
|
||||
context = await context_service.build_context(
|
||||
memory_url, depth=depth, since=since, limit=limit, offset=offset, max_related=max_related
|
||||
)
|
||||
return await to_graph_context(
|
||||
context, entity_repository=entity_repository, page=page, page_size=page_size
|
||||
)
|
||||
@@ -0,0 +1,264 @@
|
||||
"""V2 Project Router - ID-based project management operations.
|
||||
|
||||
This router provides ID-based CRUD operations for projects, replacing the
|
||||
name-based identifiers used in v1 with direct integer ID lookups.
|
||||
|
||||
Key improvements:
|
||||
- Direct database lookups via integer primary keys
|
||||
- Stable references that don't change with project renames
|
||||
- Better performance through indexed queries
|
||||
- Consistent with v2 entity operations
|
||||
"""
|
||||
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Body, Query
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.deps import (
|
||||
ProjectServiceDep,
|
||||
ProjectRepositoryDep,
|
||||
ProjectIdPathDep,
|
||||
)
|
||||
from basic_memory.schemas.project_info import (
|
||||
ProjectItem,
|
||||
ProjectStatusResponse,
|
||||
)
|
||||
from basic_memory.utils import normalize_project_path
|
||||
|
||||
router = APIRouter(prefix="/projects", tags=["project_management-v2"])
|
||||
|
||||
|
||||
@router.get("/{project_id}", response_model=ProjectItem)
|
||||
async def get_project_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
) -> ProjectItem:
|
||||
"""Get project by its numeric ID.
|
||||
|
||||
This is the primary project retrieval method in v2, using direct database
|
||||
lookups for maximum performance.
|
||||
|
||||
Args:
|
||||
project_id: Numeric project ID
|
||||
|
||||
Returns:
|
||||
Project information
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if project not found
|
||||
|
||||
Example:
|
||||
GET /v2/projects/3
|
||||
"""
|
||||
logger.info(f"API v2 request: get_project_by_id for project_id={project_id}")
|
||||
|
||||
project = await project_repository.get_by_id(project_id)
|
||||
if not project:
|
||||
raise HTTPException(status_code=404, detail=f"Project with ID {project_id} not found")
|
||||
|
||||
return ProjectItem(
|
||||
id=project.id,
|
||||
name=project.name,
|
||||
path=normalize_project_path(project.path),
|
||||
is_default=project.is_default or False,
|
||||
)
|
||||
|
||||
|
||||
@router.patch("/{project_id}", response_model=ProjectStatusResponse)
|
||||
async def update_project_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
project_service: ProjectServiceDep,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
path: Optional[str] = Body(None, description="New absolute path for the project"),
|
||||
is_active: Optional[bool] = Body(None, description="Status of the project (active/inactive)"),
|
||||
) -> ProjectStatusResponse:
|
||||
"""Update a project's information by ID.
|
||||
|
||||
Args:
|
||||
project_id: Numeric project ID
|
||||
path: Optional new absolute path for the project
|
||||
is_active: Optional status update for the project
|
||||
|
||||
Returns:
|
||||
Response confirming the project was updated
|
||||
|
||||
Raises:
|
||||
HTTPException: 400 if validation fails, 404 if project not found
|
||||
|
||||
Example:
|
||||
PATCH /v2/projects/3
|
||||
{"path": "/new/path"}
|
||||
"""
|
||||
logger.info(f"API v2 request: update_project_by_id for project_id={project_id}")
|
||||
|
||||
try:
|
||||
# Validate that path is absolute if provided
|
||||
if path and not os.path.isabs(path):
|
||||
raise HTTPException(status_code=400, detail="Path must be absolute")
|
||||
|
||||
# Get original project info for the response
|
||||
old_project = await project_repository.get_by_id(project_id)
|
||||
if not old_project:
|
||||
raise HTTPException(status_code=404, detail=f"Project with ID {project_id} not found")
|
||||
|
||||
old_project_info = ProjectItem(
|
||||
id=old_project.id,
|
||||
name=old_project.name,
|
||||
path=old_project.path,
|
||||
is_default=old_project.is_default or False,
|
||||
)
|
||||
|
||||
# Update using project name (service layer still uses names internally)
|
||||
if path:
|
||||
await project_service.move_project(old_project.name, path)
|
||||
elif is_active is not None:
|
||||
await project_service.update_project(old_project.name, is_active=is_active)
|
||||
|
||||
# Get updated project info
|
||||
updated_project = await project_repository.get_by_id(project_id)
|
||||
if not updated_project:
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"Project with ID {project_id} not found after update"
|
||||
)
|
||||
|
||||
return ProjectStatusResponse(
|
||||
message=f"Project '{updated_project.name}' updated successfully",
|
||||
status="success",
|
||||
default=(old_project.name == project_service.default_project),
|
||||
old_project=old_project_info,
|
||||
new_project=ProjectItem(
|
||||
id=updated_project.id,
|
||||
name=updated_project.name,
|
||||
path=updated_project.path,
|
||||
is_default=updated_project.is_default or False,
|
||||
),
|
||||
)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
|
||||
|
||||
@router.delete("/{project_id}", response_model=ProjectStatusResponse)
|
||||
async def delete_project_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
project_service: ProjectServiceDep,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
delete_notes: bool = Query(
|
||||
False, description="If True, delete project directory from filesystem"
|
||||
),
|
||||
) -> ProjectStatusResponse:
|
||||
"""Delete a project by ID.
|
||||
|
||||
Args:
|
||||
project_id: Numeric project ID
|
||||
delete_notes: If True, delete the project directory from the filesystem
|
||||
|
||||
Returns:
|
||||
Response confirming the project was deleted
|
||||
|
||||
Raises:
|
||||
HTTPException: 400 if trying to delete default project, 404 if not found
|
||||
|
||||
Example:
|
||||
DELETE /v2/projects/3?delete_notes=false
|
||||
"""
|
||||
logger.info(
|
||||
f"API v2 request: delete_project_by_id for project_id={project_id}, delete_notes={delete_notes}"
|
||||
)
|
||||
|
||||
try:
|
||||
old_project = await project_repository.get_by_id(project_id)
|
||||
if not old_project:
|
||||
raise HTTPException(status_code=404, detail=f"Project with ID {project_id} not found")
|
||||
|
||||
# Check if trying to delete the default project
|
||||
if old_project.name == project_service.default_project:
|
||||
available_projects = await project_service.list_projects()
|
||||
other_projects = [p.name for p in available_projects if p.id != project_id]
|
||||
detail = f"Cannot delete default project '{old_project.name}'. "
|
||||
if other_projects:
|
||||
detail += (
|
||||
f"Set another project as default first. Available: {', '.join(other_projects)}"
|
||||
)
|
||||
else:
|
||||
detail += "This is the only project in your configuration."
|
||||
raise HTTPException(status_code=400, detail=detail)
|
||||
|
||||
# Delete using project name (service layer still uses names internally)
|
||||
await project_service.remove_project(old_project.name, delete_notes=delete_notes)
|
||||
|
||||
return ProjectStatusResponse(
|
||||
message=f"Project '{old_project.name}' removed successfully",
|
||||
status="success",
|
||||
default=False,
|
||||
old_project=ProjectItem(
|
||||
id=old_project.id,
|
||||
name=old_project.name,
|
||||
path=old_project.path,
|
||||
is_default=old_project.is_default or False,
|
||||
),
|
||||
new_project=None,
|
||||
)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
|
||||
|
||||
@router.put("/{project_id}/default", response_model=ProjectStatusResponse)
|
||||
async def set_default_project_by_id(
|
||||
project_id: ProjectIdPathDep,
|
||||
project_service: ProjectServiceDep,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
) -> ProjectStatusResponse:
|
||||
"""Set a project as the default project by ID.
|
||||
|
||||
Args:
|
||||
project_id: Numeric project ID to set as default
|
||||
|
||||
Returns:
|
||||
Response confirming the project was set as default
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if project not found
|
||||
|
||||
Example:
|
||||
PUT /v2/projects/3/default
|
||||
"""
|
||||
logger.info(f"API v2 request: set_default_project_by_id for project_id={project_id}")
|
||||
|
||||
try:
|
||||
# Get the old default project
|
||||
default_name = project_service.default_project
|
||||
default_project = await project_service.get_project(default_name)
|
||||
if not default_project:
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"Default Project: '{default_name}' does not exist"
|
||||
)
|
||||
|
||||
# Get the new default project
|
||||
new_default_project = await project_repository.get_by_id(project_id)
|
||||
if not new_default_project:
|
||||
raise HTTPException(status_code=404, detail=f"Project with ID {project_id} not found")
|
||||
|
||||
# Set as default using project name (service layer still uses names internally)
|
||||
await project_service.set_default_project(new_default_project.name)
|
||||
|
||||
return ProjectStatusResponse(
|
||||
message=f"Project '{new_default_project.name}' set as default successfully",
|
||||
status="success",
|
||||
default=True,
|
||||
old_project=ProjectItem(
|
||||
id=default_project.id,
|
||||
name=default_name,
|
||||
path=default_project.path,
|
||||
is_default=False,
|
||||
),
|
||||
new_project=ProjectItem(
|
||||
id=new_default_project.id,
|
||||
name=new_default_project.name,
|
||||
path=new_default_project.path,
|
||||
is_default=True,
|
||||
),
|
||||
)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
@@ -0,0 +1,270 @@
|
||||
"""V2 Prompt Router - ID-based prompt generation operations.
|
||||
|
||||
This router uses v2 dependencies for consistent project ID handling.
|
||||
Prompt endpoints are action-based (not resource-based), so they don't
|
||||
have entity IDs in URLs - they generate formatted prompts from queries.
|
||||
"""
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from fastapi import APIRouter, HTTPException, status
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.api.routers.utils import to_graph_context, to_search_results
|
||||
from basic_memory.api.template_loader import template_loader
|
||||
from basic_memory.schemas.base import parse_timeframe
|
||||
from basic_memory.deps import (
|
||||
ContextServiceV2Dep,
|
||||
EntityRepositoryV2Dep,
|
||||
SearchServiceV2Dep,
|
||||
EntityServiceV2Dep,
|
||||
ProjectIdPathDep,
|
||||
)
|
||||
from basic_memory.schemas.prompt import (
|
||||
ContinueConversationRequest,
|
||||
SearchPromptRequest,
|
||||
PromptResponse,
|
||||
PromptMetadata,
|
||||
)
|
||||
from basic_memory.schemas.search import SearchItemType, SearchQuery
|
||||
|
||||
router = APIRouter(prefix="/prompt", tags=["prompt-v2"])
|
||||
|
||||
|
||||
@router.post("/continue-conversation", response_model=PromptResponse)
|
||||
async def continue_conversation(
|
||||
project_id: ProjectIdPathDep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
context_service: ContextServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
request: ContinueConversationRequest,
|
||||
) -> PromptResponse:
|
||||
"""Generate a prompt for continuing a conversation.
|
||||
|
||||
This endpoint takes a topic and/or timeframe and generates a prompt with
|
||||
relevant context from the knowledge base.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
request: The request parameters
|
||||
|
||||
Returns:
|
||||
Formatted continuation prompt with context
|
||||
"""
|
||||
logger.info(
|
||||
f"V2 Generating continue conversation prompt for project {project_id}, "
|
||||
f"topic: {request.topic}, timeframe: {request.timeframe}"
|
||||
)
|
||||
|
||||
since = parse_timeframe(request.timeframe) if request.timeframe else None
|
||||
|
||||
# Initialize search results
|
||||
search_results = []
|
||||
|
||||
# Get data needed for template
|
||||
if request.topic:
|
||||
query = SearchQuery(text=request.topic, after_date=request.timeframe)
|
||||
results = await search_service.search(query, limit=request.search_items_limit)
|
||||
search_results = await to_search_results(entity_service, results)
|
||||
|
||||
# Build context from results
|
||||
all_hierarchical_results = []
|
||||
for result in search_results:
|
||||
if hasattr(result, "permalink") and result.permalink:
|
||||
# Get hierarchical context using the new dataclass-based approach
|
||||
context_result = await context_service.build_context(
|
||||
result.permalink,
|
||||
depth=request.depth,
|
||||
since=since,
|
||||
max_related=request.related_items_limit,
|
||||
include_observations=True, # Include observations for entities
|
||||
)
|
||||
|
||||
# Process results into the schema format
|
||||
graph_context = await to_graph_context(
|
||||
context_result, entity_repository=entity_repository
|
||||
)
|
||||
|
||||
# Add results to our collection (limit to top results for each permalink)
|
||||
if graph_context.results:
|
||||
all_hierarchical_results.extend(graph_context.results[:3])
|
||||
|
||||
# Limit to a reasonable number of total results
|
||||
all_hierarchical_results = all_hierarchical_results[:10]
|
||||
|
||||
template_context = {
|
||||
"topic": request.topic,
|
||||
"timeframe": request.timeframe,
|
||||
"hierarchical_results": all_hierarchical_results,
|
||||
"has_results": len(all_hierarchical_results) > 0,
|
||||
}
|
||||
else:
|
||||
# If no topic, get recent activity
|
||||
context_result = await context_service.build_context(
|
||||
types=[SearchItemType.ENTITY],
|
||||
depth=request.depth,
|
||||
since=since,
|
||||
max_related=request.related_items_limit,
|
||||
include_observations=True,
|
||||
)
|
||||
recent_context = await to_graph_context(context_result, entity_repository=entity_repository)
|
||||
|
||||
hierarchical_results = recent_context.results[:5] # Limit to top 5 recent items
|
||||
|
||||
template_context = {
|
||||
"topic": f"Recent Activity from ({request.timeframe})",
|
||||
"timeframe": request.timeframe,
|
||||
"hierarchical_results": hierarchical_results,
|
||||
"has_results": len(hierarchical_results) > 0,
|
||||
}
|
||||
|
||||
try:
|
||||
# Render template
|
||||
rendered_prompt = await template_loader.render(
|
||||
"prompts/continue_conversation.hbs", template_context
|
||||
)
|
||||
|
||||
# Calculate metadata
|
||||
# Count items of different types
|
||||
observation_count = 0
|
||||
relation_count = 0
|
||||
entity_count = 0
|
||||
|
||||
# Get the hierarchical results from the template context
|
||||
hierarchical_results_for_count = template_context.get("hierarchical_results", [])
|
||||
|
||||
# For topic-based search
|
||||
if request.topic:
|
||||
for item in hierarchical_results_for_count:
|
||||
if hasattr(item, "observations"):
|
||||
observation_count += len(item.observations) if item.observations else 0
|
||||
|
||||
if hasattr(item, "related_results"):
|
||||
for related in item.related_results or []:
|
||||
if hasattr(related, "type"):
|
||||
if related.type == "relation":
|
||||
relation_count += 1
|
||||
elif related.type == "entity": # pragma: no cover
|
||||
entity_count += 1 # pragma: no cover
|
||||
# For recent activity
|
||||
else:
|
||||
for item in hierarchical_results_for_count:
|
||||
if hasattr(item, "observations"):
|
||||
observation_count += len(item.observations) if item.observations else 0
|
||||
|
||||
if hasattr(item, "related_results"):
|
||||
for related in item.related_results or []:
|
||||
if hasattr(related, "type"):
|
||||
if related.type == "relation":
|
||||
relation_count += 1
|
||||
elif related.type == "entity": # pragma: no cover
|
||||
entity_count += 1 # pragma: no cover
|
||||
|
||||
# Build metadata
|
||||
metadata = {
|
||||
"query": request.topic,
|
||||
"timeframe": request.timeframe,
|
||||
"search_count": len(search_results)
|
||||
if request.topic
|
||||
else 0, # Original search results count
|
||||
"context_count": len(hierarchical_results_for_count),
|
||||
"observation_count": observation_count,
|
||||
"relation_count": relation_count,
|
||||
"total_items": (
|
||||
len(hierarchical_results_for_count)
|
||||
+ observation_count
|
||||
+ relation_count
|
||||
+ entity_count
|
||||
),
|
||||
"search_limit": request.search_items_limit,
|
||||
"context_depth": request.depth,
|
||||
"related_limit": request.related_items_limit,
|
||||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||||
}
|
||||
|
||||
prompt_metadata = PromptMetadata(**metadata)
|
||||
|
||||
return PromptResponse(
|
||||
prompt=rendered_prompt, context=template_context, metadata=prompt_metadata
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Error rendering continue conversation template: {e}")
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=f"Error rendering prompt template: {str(e)}",
|
||||
)
|
||||
|
||||
|
||||
@router.post("/search", response_model=PromptResponse)
|
||||
async def search_prompt(
|
||||
project_id: ProjectIdPathDep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
request: SearchPromptRequest,
|
||||
page: int = 1,
|
||||
page_size: int = 10,
|
||||
) -> PromptResponse:
|
||||
"""Generate a prompt for search results.
|
||||
|
||||
This endpoint takes a search query and formats the results into a helpful
|
||||
prompt with context and suggestions.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
request: The search parameters
|
||||
page: The page number for pagination
|
||||
page_size: The number of results per page, defaults to 10
|
||||
|
||||
Returns:
|
||||
Formatted search results prompt with context
|
||||
"""
|
||||
logger.info(
|
||||
f"V2 Generating search prompt for project {project_id}, "
|
||||
f"query: {request.query}, timeframe: {request.timeframe}"
|
||||
)
|
||||
|
||||
limit = page_size
|
||||
offset = (page - 1) * page_size
|
||||
|
||||
query = SearchQuery(text=request.query, after_date=request.timeframe)
|
||||
results = await search_service.search(query, limit=limit, offset=offset)
|
||||
search_results = await to_search_results(entity_service, results)
|
||||
|
||||
template_context = {
|
||||
"query": request.query,
|
||||
"timeframe": request.timeframe,
|
||||
"results": search_results,
|
||||
"has_results": len(search_results) > 0,
|
||||
"result_count": len(search_results),
|
||||
}
|
||||
|
||||
try:
|
||||
# Render template
|
||||
rendered_prompt = await template_loader.render("prompts/search.hbs", template_context)
|
||||
|
||||
# Build metadata
|
||||
metadata = {
|
||||
"query": request.query,
|
||||
"timeframe": request.timeframe,
|
||||
"search_count": len(search_results),
|
||||
"context_count": len(search_results),
|
||||
"observation_count": 0, # Search results don't include observations
|
||||
"relation_count": 0, # Search results don't include relations
|
||||
"total_items": len(search_results),
|
||||
"search_limit": limit,
|
||||
"context_depth": 0, # No context depth for basic search
|
||||
"related_limit": 0, # No related items for basic search
|
||||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||||
}
|
||||
|
||||
prompt_metadata = PromptMetadata(**metadata)
|
||||
|
||||
return PromptResponse(
|
||||
prompt=rendered_prompt, context=template_context, metadata=prompt_metadata
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Error rendering search template: {e}")
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
|
||||
detail=f"Error rendering prompt template: {str(e)}",
|
||||
)
|
||||
@@ -0,0 +1,286 @@
|
||||
"""V2 Resource Router - ID-based resource content operations.
|
||||
|
||||
This router uses entity IDs for all operations, with file paths in request bodies
|
||||
when needed. This is consistent with v2's ID-first design.
|
||||
|
||||
Key differences from v1:
|
||||
- Uses integer entity IDs in URL paths instead of file paths
|
||||
- File paths are in request bodies for create/update operations
|
||||
- More RESTful: POST for create, PUT for update, GET for read
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Response
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.deps import (
|
||||
ProjectConfigV2Dep,
|
||||
EntityServiceV2Dep,
|
||||
FileServiceV2Dep,
|
||||
EntityRepositoryV2Dep,
|
||||
SearchServiceV2Dep,
|
||||
ProjectIdPathDep,
|
||||
)
|
||||
from basic_memory.models.knowledge import Entity as EntityModel
|
||||
from basic_memory.schemas.v2.resource import (
|
||||
CreateResourceRequest,
|
||||
UpdateResourceRequest,
|
||||
ResourceResponse,
|
||||
)
|
||||
from basic_memory.utils import validate_project_path
|
||||
|
||||
router = APIRouter(prefix="/resource", tags=["resources-v2"])
|
||||
|
||||
|
||||
@router.get("/{entity_id}")
|
||||
async def get_resource_content(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
config: ProjectConfigV2Dep,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
) -> Response:
|
||||
"""Get raw resource content by entity ID.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
entity_id: Numeric entity ID
|
||||
config: Project configuration
|
||||
entity_service: Entity service for fetching entity data
|
||||
file_service: File service for reading file content
|
||||
|
||||
Returns:
|
||||
Response with entity content
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if entity or file not found
|
||||
"""
|
||||
logger.debug(f"V2 Getting content for project {project_id}, entity_id: {entity_id}")
|
||||
|
||||
# Get entity by ID
|
||||
entities = await entity_service.get_entities_by_id([entity_id])
|
||||
if not entities:
|
||||
raise HTTPException(status_code=404, detail=f"Entity {entity_id} not found")
|
||||
|
||||
entity = entities[0]
|
||||
|
||||
# Validate entity file path to prevent path traversal
|
||||
project_path = Path(config.home)
|
||||
if not validate_project_path(entity.file_path, project_path):
|
||||
logger.error(f"Invalid file path in entity {entity.id}: {entity.file_path}")
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail="Entity contains invalid file path",
|
||||
)
|
||||
|
||||
# Check file exists via file_service (for cloud compatibility)
|
||||
if not await file_service.exists(entity.file_path):
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"File not found: {entity.file_path}",
|
||||
)
|
||||
|
||||
# Read content via file_service as bytes (works with both local and S3)
|
||||
content = await file_service.read_file_bytes(entity.file_path)
|
||||
content_type = file_service.content_type(entity.file_path)
|
||||
|
||||
return Response(content=content, media_type=content_type)
|
||||
|
||||
|
||||
@router.post("", response_model=ResourceResponse)
|
||||
async def create_resource(
|
||||
project_id: ProjectIdPathDep,
|
||||
data: CreateResourceRequest,
|
||||
config: ProjectConfigV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
) -> ResourceResponse:
|
||||
"""Create a new resource file.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
data: Create resource request with file_path and content
|
||||
config: Project configuration
|
||||
file_service: File service for writing files
|
||||
entity_repository: Entity repository for creating entities
|
||||
search_service: Search service for indexing
|
||||
|
||||
Returns:
|
||||
ResourceResponse with file information including entity_id
|
||||
|
||||
Raises:
|
||||
HTTPException: 400 for invalid file paths, 409 if file already exists
|
||||
"""
|
||||
try:
|
||||
# Validate path to prevent path traversal attacks
|
||||
project_path = Path(config.home)
|
||||
if not validate_project_path(data.file_path, project_path):
|
||||
logger.warning(
|
||||
f"Invalid file path attempted: {data.file_path} in project {config.name}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Invalid file path: {data.file_path}. "
|
||||
"Path must be relative and stay within project boundaries.",
|
||||
)
|
||||
|
||||
# Check if entity already exists
|
||||
existing_entity = await entity_repository.get_by_file_path(data.file_path)
|
||||
if existing_entity:
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=f"Resource already exists at {data.file_path} with entity_id {existing_entity.id}. "
|
||||
f"Use PUT /resource/{existing_entity.id} to update it.",
|
||||
)
|
||||
|
||||
# Cloud compatibility: avoid assuming a local filesystem path.
|
||||
# Delegate directory creation + writes to FileService (local or S3).
|
||||
await file_service.ensure_directory(Path(data.file_path).parent)
|
||||
checksum = await file_service.write_file(data.file_path, data.content)
|
||||
|
||||
# Get file info
|
||||
file_metadata = await file_service.get_file_metadata(data.file_path)
|
||||
|
||||
# Determine file details
|
||||
file_name = Path(data.file_path).name
|
||||
content_type = file_service.content_type(data.file_path)
|
||||
entity_type = "canvas" if data.file_path.endswith(".canvas") else "file"
|
||||
|
||||
# Create a new entity model
|
||||
entity = EntityModel(
|
||||
title=file_name,
|
||||
entity_type=entity_type,
|
||||
content_type=content_type,
|
||||
file_path=data.file_path,
|
||||
checksum=checksum,
|
||||
created_at=file_metadata.created_at,
|
||||
updated_at=file_metadata.modified_at,
|
||||
)
|
||||
entity = await entity_repository.add(entity)
|
||||
|
||||
# Index the file for search
|
||||
await search_service.index_entity(entity) # pyright: ignore
|
||||
|
||||
# Return success response
|
||||
return ResourceResponse(
|
||||
entity_id=entity.id,
|
||||
file_path=data.file_path,
|
||||
checksum=checksum,
|
||||
size=file_metadata.size,
|
||||
created_at=file_metadata.created_at.timestamp(),
|
||||
modified_at=file_metadata.modified_at.timestamp(),
|
||||
)
|
||||
except HTTPException:
|
||||
# Re-raise HTTP exceptions without wrapping
|
||||
raise
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.error(f"Error creating resource {data.file_path}: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Failed to create resource: {str(e)}")
|
||||
|
||||
|
||||
@router.put("/{entity_id}", response_model=ResourceResponse)
|
||||
async def update_resource(
|
||||
project_id: ProjectIdPathDep,
|
||||
entity_id: int,
|
||||
data: UpdateResourceRequest,
|
||||
config: ProjectConfigV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
) -> ResourceResponse:
|
||||
"""Update an existing resource by entity ID.
|
||||
|
||||
Can update content and optionally move the file to a new path.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
entity_id: Entity ID of the resource to update
|
||||
data: Update resource request with content and optional new file_path
|
||||
config: Project configuration
|
||||
file_service: File service for writing files
|
||||
entity_repository: Entity repository for updating entities
|
||||
search_service: Search service for indexing
|
||||
|
||||
Returns:
|
||||
ResourceResponse with updated file information
|
||||
|
||||
Raises:
|
||||
HTTPException: 404 if entity not found, 400 for invalid paths
|
||||
"""
|
||||
try:
|
||||
# Get existing entity
|
||||
entity = await entity_repository.get_by_id(entity_id)
|
||||
if not entity:
|
||||
raise HTTPException(status_code=404, detail=f"Entity {entity_id} not found")
|
||||
|
||||
# Determine target file path
|
||||
target_file_path = data.file_path if data.file_path else entity.file_path
|
||||
|
||||
# Validate path to prevent path traversal attacks
|
||||
project_path = Path(config.home)
|
||||
if not validate_project_path(target_file_path, project_path):
|
||||
logger.warning(
|
||||
f"Invalid file path attempted: {target_file_path} in project {config.name}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Invalid file path: {target_file_path}. "
|
||||
"Path must be relative and stay within project boundaries.",
|
||||
)
|
||||
|
||||
# If moving file, handle the move
|
||||
if data.file_path and data.file_path != entity.file_path:
|
||||
# Ensure new parent directory exists (no-op for S3)
|
||||
await file_service.ensure_directory(Path(target_file_path).parent)
|
||||
|
||||
# If old file exists, remove it via file_service (for cloud compatibility)
|
||||
if await file_service.exists(entity.file_path):
|
||||
await file_service.delete_file(entity.file_path)
|
||||
else:
|
||||
# Ensure directory exists for in-place update
|
||||
await file_service.ensure_directory(Path(target_file_path).parent)
|
||||
|
||||
# Write content to target file
|
||||
checksum = await file_service.write_file(target_file_path, data.content)
|
||||
|
||||
# Get file info
|
||||
file_metadata = await file_service.get_file_metadata(target_file_path)
|
||||
|
||||
# Determine file details
|
||||
file_name = Path(target_file_path).name
|
||||
content_type = file_service.content_type(target_file_path)
|
||||
entity_type = "canvas" if target_file_path.endswith(".canvas") else "file"
|
||||
|
||||
# Update entity
|
||||
updated_entity = await entity_repository.update(
|
||||
entity_id,
|
||||
{
|
||||
"title": file_name,
|
||||
"entity_type": entity_type,
|
||||
"content_type": content_type,
|
||||
"file_path": target_file_path,
|
||||
"checksum": checksum,
|
||||
"updated_at": file_metadata.modified_at,
|
||||
},
|
||||
)
|
||||
|
||||
# Index the updated file for search
|
||||
await search_service.index_entity(updated_entity) # pyright: ignore
|
||||
|
||||
# Return success response
|
||||
return ResourceResponse(
|
||||
entity_id=entity_id,
|
||||
file_path=target_file_path,
|
||||
checksum=checksum,
|
||||
size=file_metadata.size,
|
||||
created_at=file_metadata.created_at.timestamp(),
|
||||
modified_at=file_metadata.modified_at.timestamp(),
|
||||
)
|
||||
except HTTPException:
|
||||
# Re-raise HTTP exceptions without wrapping
|
||||
raise
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.error(f"Error updating resource {entity_id}: {e}")
|
||||
raise HTTPException(status_code=500, detail=f"Failed to update resource: {str(e)}")
|
||||
@@ -0,0 +1,73 @@
|
||||
"""V2 router for search operations.
|
||||
|
||||
This router uses integer project IDs for stable, efficient routing.
|
||||
V1 uses string-based project names which are less efficient and less stable.
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, BackgroundTasks
|
||||
|
||||
from basic_memory.api.routers.utils import to_search_results
|
||||
from basic_memory.schemas.search import SearchQuery, SearchResponse
|
||||
from basic_memory.deps import SearchServiceV2Dep, EntityServiceV2Dep, ProjectIdPathDep
|
||||
|
||||
# Note: No prefix here - it's added during registration as /v2/{project_id}/search
|
||||
router = APIRouter(tags=["search"])
|
||||
|
||||
|
||||
@router.post("/search/", response_model=SearchResponse)
|
||||
async def search(
|
||||
project_id: ProjectIdPathDep,
|
||||
query: SearchQuery,
|
||||
search_service: SearchServiceV2Dep,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
page: int = 1,
|
||||
page_size: int = 10,
|
||||
):
|
||||
"""Search across all knowledge and documents in a project.
|
||||
|
||||
V2 uses integer project IDs for improved performance and stability.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
query: Search query parameters (text, filters, etc.)
|
||||
search_service: Search service scoped to project
|
||||
entity_service: Entity service scoped to project
|
||||
page: Page number for pagination
|
||||
page_size: Number of results per page
|
||||
|
||||
Returns:
|
||||
SearchResponse with paginated search results
|
||||
"""
|
||||
limit = page_size
|
||||
offset = (page - 1) * page_size
|
||||
results = await search_service.search(query, limit=limit, offset=offset)
|
||||
search_results = await to_search_results(entity_service, results)
|
||||
return SearchResponse(
|
||||
results=search_results,
|
||||
current_page=page,
|
||||
page_size=page_size,
|
||||
)
|
||||
|
||||
|
||||
@router.post("/search/reindex")
|
||||
async def reindex(
|
||||
project_id: ProjectIdPathDep,
|
||||
background_tasks: BackgroundTasks,
|
||||
search_service: SearchServiceV2Dep,
|
||||
):
|
||||
"""Recreate and populate the search index for a project.
|
||||
|
||||
This is a background operation that rebuilds the search index
|
||||
from scratch. Useful after bulk updates or if the index becomes
|
||||
corrupted.
|
||||
|
||||
Args:
|
||||
project_id: Validated numeric project ID from URL path
|
||||
background_tasks: FastAPI background tasks handler
|
||||
search_service: Search service scoped to project
|
||||
|
||||
Returns:
|
||||
Status message indicating reindex has been initiated
|
||||
"""
|
||||
await search_service.reindex_all(background_tasks=background_tasks)
|
||||
return {"status": "ok", "message": "Reindex initiated"}
|
||||
@@ -2,7 +2,7 @@ from typing import Optional
|
||||
|
||||
import typer
|
||||
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.config import ConfigManager, init_cli_logging
|
||||
|
||||
|
||||
def version_callback(value: bool) -> None:
|
||||
@@ -31,6 +31,9 @@ def app_callback(
|
||||
) -> None:
|
||||
"""Basic Memory - Local-first personal knowledge management."""
|
||||
|
||||
# Initialize logging for CLI (file only, no stdout)
|
||||
init_cli_logging()
|
||||
|
||||
# Run initialization for every command unless --version was specified
|
||||
if not version and ctx.invoked_subcommand is not None:
|
||||
from basic_memory.services.initialization import ensure_initialization
|
||||
|
||||
@@ -16,6 +16,7 @@ from typing import Optional
|
||||
|
||||
from rich.console import Console
|
||||
|
||||
from basic_memory.cli.commands.cloud.rclone_installer import is_rclone_installed
|
||||
from basic_memory.utils import normalize_project_path
|
||||
|
||||
console = Console()
|
||||
@@ -27,6 +28,21 @@ class RcloneError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def check_rclone_installed() -> None:
|
||||
"""Check if rclone is installed and raise helpful error if not.
|
||||
|
||||
Raises:
|
||||
RcloneError: If rclone is not installed with installation instructions
|
||||
"""
|
||||
if not is_rclone_installed():
|
||||
raise RcloneError(
|
||||
"rclone is not installed.\n\n"
|
||||
"Install rclone by running: bm cloud setup\n"
|
||||
"Or install manually from: https://rclone.org/downloads/\n\n"
|
||||
"Windows users: Ensure you have a package manager installed (winget, chocolatey, or scoop)"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SyncProject:
|
||||
"""Project configured for cloud sync.
|
||||
@@ -124,8 +140,10 @@ def project_sync(
|
||||
True if sync succeeded, False otherwise
|
||||
|
||||
Raises:
|
||||
RcloneError: If project has no local_sync_path configured
|
||||
RcloneError: If project has no local_sync_path configured or rclone not installed
|
||||
"""
|
||||
check_rclone_installed()
|
||||
|
||||
if not project.local_sync_path:
|
||||
raise RcloneError(f"Project {project.name} has no local_sync_path configured")
|
||||
|
||||
@@ -180,8 +198,10 @@ def project_bisync(
|
||||
True if bisync succeeded, False otherwise
|
||||
|
||||
Raises:
|
||||
RcloneError: If project has no local_sync_path or needs --resync
|
||||
RcloneError: If project has no local_sync_path, needs --resync, or rclone not installed
|
||||
"""
|
||||
check_rclone_installed()
|
||||
|
||||
if not project.local_sync_path:
|
||||
raise RcloneError(f"Project {project.name} has no local_sync_path configured")
|
||||
|
||||
@@ -249,8 +269,10 @@ def project_check(
|
||||
True if files match, False if differences found
|
||||
|
||||
Raises:
|
||||
RcloneError: If project has no local_sync_path configured
|
||||
RcloneError: If project has no local_sync_path configured or rclone not installed
|
||||
"""
|
||||
check_rclone_installed()
|
||||
|
||||
if not project.local_sync_path:
|
||||
raise RcloneError(f"Project {project.name} has no local_sync_path configured")
|
||||
|
||||
@@ -291,7 +313,10 @@ def project_ls(
|
||||
|
||||
Raises:
|
||||
subprocess.CalledProcessError: If rclone command fails
|
||||
RcloneError: If rclone is not installed
|
||||
"""
|
||||
check_rclone_installed()
|
||||
|
||||
remote_path = get_project_remote(project, bucket_name)
|
||||
if path:
|
||||
remote_path = f"{remote_path}/{path}"
|
||||
|
||||
@@ -151,11 +151,25 @@ def install_rclone_windows() -> None:
|
||||
except RcloneInstallError:
|
||||
console.print("[yellow]scoop installation failed[/yellow]")
|
||||
|
||||
# No package manager available
|
||||
raise RcloneInstallError(
|
||||
"Could not install rclone automatically. Please install a package manager "
|
||||
"(winget, chocolatey, or scoop) or install rclone manually from https://rclone.org/downloads/"
|
||||
# No package manager available - provide detailed instructions
|
||||
error_msg = (
|
||||
"Could not install rclone automatically.\n\n"
|
||||
"Windows requires a package manager to install rclone. Options:\n\n"
|
||||
"1. Install winget (recommended, built into Windows 11):\n"
|
||||
" - Windows 11: Already installed\n"
|
||||
" - Windows 10: Install 'App Installer' from Microsoft Store\n"
|
||||
" - Then run: bm cloud setup\n\n"
|
||||
"2. Install chocolatey:\n"
|
||||
" - Visit: https://chocolatey.org/install\n"
|
||||
" - Then run: bm cloud setup\n\n"
|
||||
"3. Install scoop:\n"
|
||||
" - Visit: https://scoop.sh\n"
|
||||
" - Then run: bm cloud setup\n\n"
|
||||
"4. Manual installation:\n"
|
||||
" - Download from: https://rclone.org/downloads/\n"
|
||||
" - Extract and add to PATH\n"
|
||||
)
|
||||
raise RcloneInstallError(error_msg)
|
||||
|
||||
|
||||
def install_rclone(platform_override: Optional[str] = None) -> None:
|
||||
|
||||
@@ -6,7 +6,7 @@ import typer
|
||||
from typing import Optional
|
||||
|
||||
from basic_memory.cli.app import app
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.config import ConfigManager, init_mcp_logging
|
||||
|
||||
# Import mcp instance
|
||||
from basic_memory.mcp.server import mcp as mcp_server # pragma: no cover
|
||||
@@ -44,6 +44,8 @@ if not config.cloud_mode_enabled:
|
||||
- streamable-http: Recommended for web deployments (default)
|
||||
- sse: Server-Sent Events (for compatibility with existing clients)
|
||||
"""
|
||||
# Initialize logging for MCP (file only, stdout breaks protocol)
|
||||
init_mcp_logging()
|
||||
|
||||
# Validate and set project constraint if specified
|
||||
if project:
|
||||
|
||||
+131
-81
@@ -6,12 +6,12 @@ from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Literal, Optional, List, Tuple
|
||||
from enum import Enum
|
||||
|
||||
from loguru import logger
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
from pydantic import BaseModel, Field, model_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
import basic_memory
|
||||
from basic_memory.utils import setup_logging, generate_permalink
|
||||
|
||||
|
||||
@@ -24,6 +24,13 @@ WATCH_STATUS_JSON = "watch-status.json"
|
||||
Environment = Literal["test", "dev", "user"]
|
||||
|
||||
|
||||
class DatabaseBackend(str, Enum):
|
||||
"""Supported database backends."""
|
||||
|
||||
SQLITE = "sqlite"
|
||||
POSTGRES = "postgres"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ProjectConfig:
|
||||
"""Configuration for a specific basic-memory project."""
|
||||
@@ -63,8 +70,10 @@ class BasicMemoryConfig(BaseSettings):
|
||||
|
||||
projects: Dict[str, str] = Field(
|
||||
default_factory=lambda: {
|
||||
"main": Path(os.getenv("BASIC_MEMORY_HOME", Path.home() / "basic-memory")).as_posix()
|
||||
},
|
||||
"main": str(Path(os.getenv("BASIC_MEMORY_HOME", Path.home() / "basic-memory")))
|
||||
}
|
||||
if os.getenv("BASIC_MEMORY_HOME")
|
||||
else {},
|
||||
description="Mapping of project names to their filesystem paths",
|
||||
)
|
||||
default_project: str = Field(
|
||||
@@ -79,13 +88,43 @@ class BasicMemoryConfig(BaseSettings):
|
||||
# overridden by ~/.basic-memory/config.json
|
||||
log_level: str = "INFO"
|
||||
|
||||
# Database configuration
|
||||
database_backend: DatabaseBackend = Field(
|
||||
default=DatabaseBackend.SQLITE,
|
||||
description="Database backend to use (sqlite or postgres)",
|
||||
)
|
||||
|
||||
database_url: Optional[str] = Field(
|
||||
default=None,
|
||||
description="Database connection URL. For Postgres, use postgresql+asyncpg://user:pass@host:port/db. If not set, SQLite will use default path.",
|
||||
)
|
||||
|
||||
# Database connection pool configuration (Postgres only)
|
||||
db_pool_size: int = Field(
|
||||
default=20,
|
||||
description="Number of connections to keep in the pool (Postgres only)",
|
||||
gt=0,
|
||||
)
|
||||
db_pool_overflow: int = Field(
|
||||
default=40,
|
||||
description="Max additional connections beyond pool_size under load (Postgres only)",
|
||||
gt=0,
|
||||
)
|
||||
db_pool_recycle: int = Field(
|
||||
default=180,
|
||||
description="Recycle connections after N seconds to prevent stale connections. Default 180s works well with Neon's ~5 minute scale-to-zero (Postgres only)",
|
||||
gt=0,
|
||||
)
|
||||
|
||||
# Watch service configuration
|
||||
sync_delay: int = Field(
|
||||
default=1000, description="Milliseconds to wait after changes before syncing", gt=0
|
||||
)
|
||||
|
||||
watch_project_reload_interval: int = Field(
|
||||
default=30, description="Seconds between reloading project list in watch service", gt=0
|
||||
default=300,
|
||||
description="Seconds between reloading project list in watch service. Higher values reduce CPU usage by minimizing watcher restarts. Default 300s (5 min) balances efficiency with responsiveness to new projects.",
|
||||
gt=0,
|
||||
)
|
||||
|
||||
# update permalinks on move
|
||||
@@ -176,6 +215,36 @@ class BasicMemoryConfig(BaseSettings):
|
||||
# Fall back to config file value
|
||||
return self.cloud_mode
|
||||
|
||||
@classmethod
|
||||
def for_cloud_tenant(
|
||||
cls,
|
||||
database_url: str,
|
||||
projects: Optional[Dict[str, str]] = None,
|
||||
) -> "BasicMemoryConfig":
|
||||
"""Create config for cloud tenant - no config.json, database is source of truth.
|
||||
|
||||
This factory method creates a BasicMemoryConfig suitable for cloud deployments
|
||||
where:
|
||||
- Database is Postgres (Neon), not SQLite
|
||||
- Projects are discovered from the database, not config file
|
||||
- Path validation is skipped (no local filesystem in cloud)
|
||||
- Initialization sync is skipped (stateless deployment)
|
||||
|
||||
Args:
|
||||
database_url: Postgres connection URL for tenant database
|
||||
projects: Optional project mapping (usually empty, discovered from DB)
|
||||
|
||||
Returns:
|
||||
BasicMemoryConfig configured for cloud mode
|
||||
"""
|
||||
return cls(
|
||||
database_backend=DatabaseBackend.POSTGRES,
|
||||
database_url=database_url,
|
||||
projects=projects or {},
|
||||
cloud_mode=True,
|
||||
skip_initialization_sync=True,
|
||||
)
|
||||
|
||||
model_config = SettingsConfigDict(
|
||||
env_prefix="BASIC_MEMORY_",
|
||||
extra="ignore",
|
||||
@@ -192,15 +261,20 @@ class BasicMemoryConfig(BaseSettings):
|
||||
|
||||
def model_post_init(self, __context: Any) -> None:
|
||||
"""Ensure configuration is valid after initialization."""
|
||||
# Ensure main project exists
|
||||
if "main" not in self.projects: # pragma: no cover
|
||||
self.projects["main"] = (
|
||||
Path(os.getenv("BASIC_MEMORY_HOME", Path.home() / "basic-memory"))
|
||||
).as_posix()
|
||||
# Skip project initialization in cloud mode - projects are discovered from DB
|
||||
if self.database_backend == DatabaseBackend.POSTGRES:
|
||||
return
|
||||
|
||||
# Ensure default project is valid
|
||||
# Ensure at least one project exists; if none exist then create main
|
||||
if not self.projects: # pragma: no cover
|
||||
self.projects["main"] = str(
|
||||
Path(os.getenv("BASIC_MEMORY_HOME", Path.home() / "basic-memory"))
|
||||
)
|
||||
|
||||
# Ensure default project is valid (i.e. points to an existing project)
|
||||
if self.default_project not in self.projects: # pragma: no cover
|
||||
self.default_project = "main"
|
||||
# Set default to first available project
|
||||
self.default_project = next(iter(self.projects.keys()))
|
||||
|
||||
@property
|
||||
def app_database_path(self) -> Path:
|
||||
@@ -233,19 +307,26 @@ class BasicMemoryConfig(BaseSettings):
|
||||
"""Get all configured projects as ProjectConfig objects."""
|
||||
return [ProjectConfig(name=name, home=Path(path)) for name, path in self.projects.items()]
|
||||
|
||||
@field_validator("projects")
|
||||
@classmethod
|
||||
def ensure_project_paths_exists(cls, v: Dict[str, str]) -> Dict[str, str]: # pragma: no cover
|
||||
"""Ensure project path exists."""
|
||||
for name, path_value in v.items():
|
||||
@model_validator(mode="after")
|
||||
def ensure_project_paths_exists(self) -> "BasicMemoryConfig": # pragma: no cover
|
||||
"""Ensure project paths exist.
|
||||
|
||||
Skips path creation when using Postgres backend (cloud mode) since
|
||||
cloud tenants don't use local filesystem paths.
|
||||
"""
|
||||
# Skip path creation for cloud mode - no local filesystem
|
||||
if self.database_backend == DatabaseBackend.POSTGRES:
|
||||
return self
|
||||
|
||||
for name, path_value in self.projects.items():
|
||||
path = Path(path_value)
|
||||
if not Path(path).exists():
|
||||
if not path.exists():
|
||||
try:
|
||||
path.mkdir(parents=True)
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create project path: {e}")
|
||||
raise e
|
||||
return v
|
||||
return self
|
||||
|
||||
@property
|
||||
def data_dir_path(self):
|
||||
@@ -358,7 +439,7 @@ class ConfigManager:
|
||||
|
||||
# Load config, modify it, and save it
|
||||
config = self.load_config()
|
||||
config.projects[name] = project_path.as_posix()
|
||||
config.projects[name] = str(project_path)
|
||||
self.save_config(config)
|
||||
return ProjectConfig(name=name, home=project_path)
|
||||
|
||||
@@ -448,69 +529,38 @@ def save_basic_memory_config(file_path: Path, config: BasicMemoryConfig) -> None
|
||||
logger.error(f"Failed to save config: {e}")
|
||||
|
||||
|
||||
# setup logging to a single log file in user home directory
|
||||
user_home = Path.home()
|
||||
log_dir = user_home / DATA_DIR_NAME
|
||||
log_dir.mkdir(parents=True, exist_ok=True)
|
||||
# Logging initialization functions for different entry points
|
||||
|
||||
|
||||
# Process info for logging
|
||||
def get_process_name(): # pragma: no cover
|
||||
def init_cli_logging() -> None: # pragma: no cover
|
||||
"""Initialize logging for CLI commands - file only.
|
||||
|
||||
CLI commands should not log to stdout to avoid interfering with
|
||||
command output and shell integration.
|
||||
"""
|
||||
get the type of process for logging
|
||||
"""
|
||||
import sys
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL", "INFO")
|
||||
setup_logging(log_level=log_level, log_to_file=True)
|
||||
|
||||
if "sync" in sys.argv:
|
||||
return "sync"
|
||||
elif "mcp" in sys.argv:
|
||||
return "mcp"
|
||||
elif "cli" in sys.argv:
|
||||
return "cli"
|
||||
|
||||
def init_mcp_logging() -> None: # pragma: no cover
|
||||
"""Initialize logging for MCP server - file only.
|
||||
|
||||
MCP server must not log to stdout as it would corrupt the
|
||||
JSON-RPC protocol communication.
|
||||
"""
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL", "INFO")
|
||||
setup_logging(log_level=log_level, log_to_file=True)
|
||||
|
||||
|
||||
def init_api_logging() -> None: # pragma: no cover
|
||||
"""Initialize logging for API server.
|
||||
|
||||
Cloud mode (BASIC_MEMORY_CLOUD_MODE=1): stdout with structured context
|
||||
Local mode: file only
|
||||
"""
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL", "INFO")
|
||||
cloud_mode = os.getenv("BASIC_MEMORY_CLOUD_MODE", "").lower() in ("1", "true")
|
||||
if cloud_mode:
|
||||
setup_logging(log_level=log_level, log_to_stdout=True, structured_context=True)
|
||||
else:
|
||||
return "api"
|
||||
|
||||
|
||||
process_name = get_process_name()
|
||||
|
||||
# Global flag to track if logging has been set up
|
||||
_LOGGING_SETUP = False
|
||||
|
||||
|
||||
# Logging
|
||||
|
||||
|
||||
def setup_basic_memory_logging(): # pragma: no cover
|
||||
"""Set up logging for basic-memory, ensuring it only happens once."""
|
||||
global _LOGGING_SETUP
|
||||
if _LOGGING_SETUP:
|
||||
# We can't log before logging is set up
|
||||
# print("Skipping duplicate logging setup")
|
||||
return
|
||||
|
||||
# Check for console logging environment variable - accept more truthy values
|
||||
console_logging_env = os.getenv("BASIC_MEMORY_CONSOLE_LOGGING", "false").lower()
|
||||
console_logging = console_logging_env in ("true", "1", "yes", "on")
|
||||
|
||||
# Check for log level environment variable first, fall back to config
|
||||
log_level = os.getenv("BASIC_MEMORY_LOG_LEVEL")
|
||||
if not log_level:
|
||||
config_manager = ConfigManager()
|
||||
log_level = config_manager.config.log_level
|
||||
|
||||
config_manager = ConfigManager()
|
||||
config = get_project_config()
|
||||
setup_logging(
|
||||
env=config_manager.config.env,
|
||||
home_dir=user_home, # Use user home for logs
|
||||
log_level=log_level,
|
||||
log_file=f"{DATA_DIR_NAME}/basic-memory-{process_name}.log",
|
||||
console=console_logging,
|
||||
)
|
||||
|
||||
logger.info(f"Basic Memory {basic_memory.__version__} (Project: {config.project})")
|
||||
_LOGGING_SETUP = True
|
||||
|
||||
|
||||
# Set up logging
|
||||
setup_basic_memory_logging()
|
||||
setup_logging(log_level=log_level, log_to_file=True)
|
||||
|
||||
+139
-73
@@ -5,7 +5,7 @@ from enum import Enum, auto
|
||||
from pathlib import Path
|
||||
from typing import AsyncGenerator, Optional
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, ConfigManager
|
||||
from basic_memory.config import BasicMemoryConfig, ConfigManager, DatabaseBackend
|
||||
from alembic import command
|
||||
from alembic.config import Config
|
||||
|
||||
@@ -20,12 +20,12 @@ from sqlalchemy.ext.asyncio import (
|
||||
)
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
from basic_memory.repository.search_repository import SearchRepository
|
||||
from basic_memory.repository.postgres_search_repository import PostgresSearchRepository
|
||||
from basic_memory.repository.sqlite_search_repository import SQLiteSearchRepository
|
||||
|
||||
# Module level state
|
||||
_engine: Optional[AsyncEngine] = None
|
||||
_session_maker: Optional[async_sessionmaker[AsyncSession]] = None
|
||||
_migrations_completed: bool = False
|
||||
|
||||
|
||||
class DatabaseType(Enum):
|
||||
@@ -33,10 +33,41 @@ class DatabaseType(Enum):
|
||||
|
||||
MEMORY = auto()
|
||||
FILESYSTEM = auto()
|
||||
POSTGRES = auto()
|
||||
|
||||
@classmethod
|
||||
def get_db_url(cls, db_path: Path, db_type: "DatabaseType") -> str:
|
||||
"""Get SQLAlchemy URL for database path."""
|
||||
def get_db_url(
|
||||
cls, db_path: Path, db_type: "DatabaseType", config: Optional[BasicMemoryConfig] = None
|
||||
) -> str:
|
||||
"""Get SQLAlchemy URL for database path.
|
||||
|
||||
Args:
|
||||
db_path: Path to SQLite database file (ignored for Postgres)
|
||||
db_type: Type of database (MEMORY, FILESYSTEM, or POSTGRES)
|
||||
config: Optional config to check for database backend and URL
|
||||
|
||||
Returns:
|
||||
SQLAlchemy connection URL
|
||||
"""
|
||||
# Load config if not provided
|
||||
if config is None:
|
||||
config = ConfigManager().config
|
||||
|
||||
# Handle explicit Postgres type
|
||||
if db_type == cls.POSTGRES:
|
||||
if not config.database_url:
|
||||
raise ValueError("DATABASE_URL must be set when using Postgres backend")
|
||||
logger.info(f"Using Postgres database: {config.database_url}")
|
||||
return config.database_url
|
||||
|
||||
# Check if Postgres backend is configured (for backward compatibility)
|
||||
if config.database_backend == DatabaseBackend.POSTGRES:
|
||||
if not config.database_url:
|
||||
raise ValueError("DATABASE_URL must be set when using Postgres backend")
|
||||
logger.info(f"Using Postgres database: {config.database_url}")
|
||||
return config.database_url
|
||||
|
||||
# SQLite databases
|
||||
if db_type == cls.MEMORY:
|
||||
logger.info("Using in-memory SQLite database")
|
||||
return "sqlite+aiosqlite://"
|
||||
@@ -64,7 +95,14 @@ async def scoped_session(
|
||||
factory = get_scoped_session_factory(session_maker)
|
||||
session = factory()
|
||||
try:
|
||||
await session.execute(text("PRAGMA foreign_keys=ON"))
|
||||
# Only enable foreign keys for SQLite (Postgres has them enabled by default)
|
||||
# Detect database type from session's bind (engine) dialect
|
||||
engine = session.get_bind()
|
||||
dialect_name = engine.dialect.name
|
||||
|
||||
if dialect_name == "sqlite":
|
||||
await session.execute(text("PRAGMA foreign_keys=ON"))
|
||||
|
||||
yield session
|
||||
await session.commit()
|
||||
except Exception:
|
||||
@@ -103,13 +141,16 @@ def _configure_sqlite_connection(dbapi_conn, enable_wal: bool = True) -> None:
|
||||
cursor.close()
|
||||
|
||||
|
||||
def _create_engine_and_session(
|
||||
db_path: Path, db_type: DatabaseType = DatabaseType.FILESYSTEM
|
||||
) -> tuple[AsyncEngine, async_sessionmaker[AsyncSession]]:
|
||||
"""Internal helper to create engine and session maker."""
|
||||
db_url = DatabaseType.get_db_url(db_path, db_type)
|
||||
logger.debug(f"Creating engine for db_url: {db_url}")
|
||||
def _create_sqlite_engine(db_url: str, db_type: DatabaseType) -> AsyncEngine:
|
||||
"""Create SQLite async engine with appropriate configuration.
|
||||
|
||||
Args:
|
||||
db_url: SQLite connection URL
|
||||
db_type: Database type (MEMORY or FILESYSTEM)
|
||||
|
||||
Returns:
|
||||
Configured async engine for SQLite
|
||||
"""
|
||||
# Configure connection args with Windows-specific settings
|
||||
connect_args: dict[str, bool | float | None] = {"check_same_thread": False}
|
||||
|
||||
@@ -146,6 +187,67 @@ def _create_engine_and_session(
|
||||
"""Enable WAL mode on each connection."""
|
||||
_configure_sqlite_connection(dbapi_conn, enable_wal=enable_wal)
|
||||
|
||||
return engine
|
||||
|
||||
|
||||
def _create_postgres_engine(db_url: str, config: BasicMemoryConfig) -> AsyncEngine:
|
||||
"""Create Postgres async engine with appropriate configuration.
|
||||
|
||||
Args:
|
||||
db_url: Postgres connection URL (postgresql+asyncpg://...)
|
||||
config: BasicMemoryConfig with pool settings
|
||||
|
||||
Returns:
|
||||
Configured async engine for Postgres
|
||||
"""
|
||||
# Use NullPool connection issues.
|
||||
# Assume connection pooler like PgBouncer handles connection pooling.
|
||||
engine = create_async_engine(
|
||||
db_url,
|
||||
echo=False,
|
||||
poolclass=NullPool, # No pooling - fresh connection per request
|
||||
connect_args={
|
||||
# Disable statement cache to avoid issues with prepared statements on reconnect
|
||||
"statement_cache_size": 0,
|
||||
# Allow 30s for commands (Neon cold start can take 2-5s, sometimes longer)
|
||||
"command_timeout": 30,
|
||||
# Allow 30s for initial connection (Neon wake-up time)
|
||||
"timeout": 30,
|
||||
"server_settings": {
|
||||
"application_name": "basic-memory",
|
||||
# Statement timeout for queries (30s to allow for cold start)
|
||||
"statement_timeout": "30s",
|
||||
},
|
||||
},
|
||||
)
|
||||
logger.debug("Created Postgres engine with NullPool (no connection pooling)")
|
||||
|
||||
return engine
|
||||
|
||||
|
||||
def _create_engine_and_session(
|
||||
db_path: Path, db_type: DatabaseType = DatabaseType.FILESYSTEM
|
||||
) -> tuple[AsyncEngine, async_sessionmaker[AsyncSession]]:
|
||||
"""Internal helper to create engine and session maker.
|
||||
|
||||
Args:
|
||||
db_path: Path to database file (used for SQLite, ignored for Postgres)
|
||||
db_type: Type of database (MEMORY, FILESYSTEM, or POSTGRES)
|
||||
|
||||
Returns:
|
||||
Tuple of (engine, session_maker)
|
||||
"""
|
||||
config = ConfigManager().config
|
||||
db_url = DatabaseType.get_db_url(db_path, db_type, config)
|
||||
logger.debug(f"Creating engine for db_url: {db_url}")
|
||||
|
||||
# Delegate to backend-specific engine creation
|
||||
# Check explicit POSTGRES type first, then config setting
|
||||
if db_type == DatabaseType.POSTGRES or config.database_backend == DatabaseBackend.POSTGRES:
|
||||
engine = _create_postgres_engine(db_url, config)
|
||||
else:
|
||||
engine = _create_sqlite_engine(db_url, db_type)
|
||||
|
||||
session_maker = async_sessionmaker(engine, expire_on_commit=False)
|
||||
return engine, session_maker
|
||||
|
||||
@@ -181,13 +283,12 @@ async def get_or_create_db(
|
||||
|
||||
async def shutdown_db() -> None: # pragma: no cover
|
||||
"""Clean up database connections."""
|
||||
global _engine, _session_maker, _migrations_completed
|
||||
global _engine, _session_maker
|
||||
|
||||
if _engine:
|
||||
await _engine.dispose()
|
||||
_engine = None
|
||||
_session_maker = None
|
||||
_migrations_completed = False
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
@@ -201,50 +302,12 @@ async def engine_session_factory(
|
||||
for each test. For production use, use get_or_create_db() instead.
|
||||
"""
|
||||
|
||||
global _engine, _session_maker, _migrations_completed
|
||||
global _engine, _session_maker
|
||||
|
||||
db_url = DatabaseType.get_db_url(db_path, db_type)
|
||||
logger.debug(f"Creating engine for db_url: {db_url}")
|
||||
|
||||
# Configure connection args with Windows-specific settings
|
||||
connect_args: dict[str, bool | float | None] = {"check_same_thread": False}
|
||||
|
||||
# Add Windows-specific parameters to improve reliability
|
||||
if os.name == "nt": # Windows
|
||||
connect_args.update(
|
||||
{
|
||||
"timeout": 30.0, # Increase timeout to 30 seconds for Windows
|
||||
"isolation_level": None, # Use autocommit mode
|
||||
}
|
||||
)
|
||||
# Use NullPool for Windows filesystem databases to avoid connection pooling issues
|
||||
# Important: Do NOT use NullPool for in-memory databases as it will destroy the database
|
||||
# between connections
|
||||
if db_type == DatabaseType.FILESYSTEM:
|
||||
_engine = create_async_engine(
|
||||
db_url,
|
||||
connect_args=connect_args,
|
||||
poolclass=NullPool, # Disable connection pooling on Windows
|
||||
echo=False,
|
||||
)
|
||||
else:
|
||||
# In-memory databases need connection pooling to maintain state
|
||||
_engine = create_async_engine(db_url, connect_args=connect_args)
|
||||
else:
|
||||
_engine = create_async_engine(db_url, connect_args=connect_args)
|
||||
|
||||
# Enable WAL mode for better concurrency and reliability
|
||||
# Note: WAL mode is not supported for in-memory databases
|
||||
enable_wal = db_type != DatabaseType.MEMORY
|
||||
|
||||
@event.listens_for(_engine.sync_engine, "connect")
|
||||
def enable_wal_mode(dbapi_conn, connection_record):
|
||||
"""Enable WAL mode on each connection."""
|
||||
_configure_sqlite_connection(dbapi_conn, enable_wal=enable_wal)
|
||||
# Use the same helper function as production code
|
||||
_engine, _session_maker = _create_engine_and_session(db_path, db_type)
|
||||
|
||||
try:
|
||||
_session_maker = async_sessionmaker(_engine, expire_on_commit=False)
|
||||
|
||||
# Verify that engine and session maker are initialized
|
||||
if _engine is None: # pragma: no cover
|
||||
logger.error("Database engine is None in engine_session_factory")
|
||||
@@ -260,20 +323,16 @@ async def engine_session_factory(
|
||||
await _engine.dispose()
|
||||
_engine = None
|
||||
_session_maker = None
|
||||
_migrations_completed = False
|
||||
|
||||
|
||||
async def run_migrations(
|
||||
app_config: BasicMemoryConfig, database_type=DatabaseType.FILESYSTEM, force: bool = False
|
||||
app_config: BasicMemoryConfig, database_type=DatabaseType.FILESYSTEM
|
||||
): # pragma: no cover
|
||||
"""Run any pending alembic migrations."""
|
||||
global _migrations_completed
|
||||
|
||||
# Skip if migrations already completed unless forced
|
||||
if _migrations_completed and not force:
|
||||
logger.debug("Migrations already completed in this session, skipping")
|
||||
return
|
||||
"""Run any pending alembic migrations.
|
||||
|
||||
Note: Alembic tracks which migrations have been applied via the alembic_version table,
|
||||
so it's safe to call this multiple times - it will only run pending migrations.
|
||||
"""
|
||||
logger.info("Running database migrations...")
|
||||
try:
|
||||
# Get the absolute path to the alembic directory relative to this file
|
||||
@@ -288,9 +347,11 @@ async def run_migrations(
|
||||
)
|
||||
config.set_main_option("timezone", "UTC")
|
||||
config.set_main_option("revision_environment", "false")
|
||||
config.set_main_option(
|
||||
"sqlalchemy.url", DatabaseType.get_db_url(app_config.database_path, database_type)
|
||||
)
|
||||
|
||||
# Get the correct database URL based on backend configuration
|
||||
# No URL conversion needed - env.py now handles both async and sync engines
|
||||
db_url = DatabaseType.get_db_url(app_config.database_path, database_type, app_config)
|
||||
config.set_main_option("sqlalchemy.url", db_url)
|
||||
|
||||
command.upgrade(config, "head")
|
||||
logger.info("Migrations completed successfully")
|
||||
@@ -301,12 +362,17 @@ async def run_migrations(
|
||||
else:
|
||||
session_maker = _session_maker
|
||||
|
||||
# initialize the search Index schema
|
||||
# the project_id is not used for init_search_index, so we pass a dummy value
|
||||
await SearchRepository(session_maker, 1).init_search_index()
|
||||
|
||||
# Mark migrations as completed
|
||||
_migrations_completed = True
|
||||
# Initialize the search index schema
|
||||
# For SQLite: Create FTS5 virtual table
|
||||
# For Postgres: No-op (tsvector column added by migrations)
|
||||
# The project_id is not used for init_search_index, so we pass a dummy value
|
||||
if (
|
||||
database_type == DatabaseType.POSTGRES
|
||||
or app_config.database_backend == DatabaseBackend.POSTGRES
|
||||
):
|
||||
await PostgresSearchRepository(session_maker, 1).init_search_index()
|
||||
else:
|
||||
await SQLiteSearchRepository(session_maker, 1).init_search_index()
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.error(f"Error running migrations: {e}")
|
||||
raise
|
||||
|
||||
+289
-7
@@ -25,7 +25,7 @@ from basic_memory.repository.entity_repository import EntityRepository
|
||||
from basic_memory.repository.observation_repository import ObservationRepository
|
||||
from basic_memory.repository.project_repository import ProjectRepository
|
||||
from basic_memory.repository.relation_repository import RelationRepository
|
||||
from basic_memory.repository.search_repository import SearchRepository
|
||||
from basic_memory.repository.search_repository import SearchRepository, create_search_repository
|
||||
from basic_memory.services import EntityService, ProjectService
|
||||
from basic_memory.services.context_service import ContextService
|
||||
from basic_memory.services.directory_service import DirectoryService
|
||||
@@ -76,6 +76,34 @@ async def get_project_config(
|
||||
|
||||
ProjectConfigDep = Annotated[ProjectConfig, Depends(get_project_config)] # pragma: no cover
|
||||
|
||||
|
||||
async def get_project_config_v2(
|
||||
project_id: "ProjectIdPathDep", project_repository: "ProjectRepositoryDep"
|
||||
) -> ProjectConfig: # pragma: no cover
|
||||
"""Get the project config for v2 API (uses integer project_id from path).
|
||||
|
||||
Args:
|
||||
project_id: The validated numeric project ID from the URL path
|
||||
project_repository: Repository for project operations
|
||||
|
||||
Returns:
|
||||
The resolved project config
|
||||
|
||||
Raises:
|
||||
HTTPException: If project is not found
|
||||
"""
|
||||
project_obj = await project_repository.get_by_id(project_id)
|
||||
if project_obj:
|
||||
return ProjectConfig(name=project_obj.name, home=pathlib.Path(project_obj.path))
|
||||
|
||||
# Not found (this should not happen since ProjectIdPathDep already validates existence)
|
||||
raise HTTPException( # pragma: no cover
|
||||
status_code=status.HTTP_404_NOT_FOUND, detail=f"Project with ID {project_id} not found."
|
||||
)
|
||||
|
||||
|
||||
ProjectConfigV2Dep = Annotated[ProjectConfig, Depends(get_project_config_v2)] # pragma: no cover
|
||||
|
||||
## sqlalchemy
|
||||
|
||||
|
||||
@@ -130,6 +158,38 @@ ProjectRepositoryDep = Annotated[ProjectRepository, Depends(get_project_reposito
|
||||
ProjectPathDep = Annotated[str, Path()] # Use Path dependency to extract from URL
|
||||
|
||||
|
||||
async def validate_project_id(
|
||||
project_id: int,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
) -> int:
|
||||
"""Validate that a numeric project ID exists in the database.
|
||||
|
||||
This is used for v2 API endpoints that take project IDs as integers in the path.
|
||||
The project_id parameter will be automatically extracted from the URL path by FastAPI.
|
||||
|
||||
Args:
|
||||
project_id: The numeric project ID from the URL path
|
||||
project_repository: Repository for project operations
|
||||
|
||||
Returns:
|
||||
The validated project ID
|
||||
|
||||
Raises:
|
||||
HTTPException: If project with that ID is not found
|
||||
"""
|
||||
project_obj = await project_repository.get_by_id(project_id)
|
||||
if not project_obj:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_404_NOT_FOUND,
|
||||
detail=f"Project with ID {project_id} not found.",
|
||||
)
|
||||
return project_id
|
||||
|
||||
|
||||
# V2 API: Validated integer project ID from path
|
||||
ProjectIdPathDep = Annotated[int, Depends(validate_project_id)]
|
||||
|
||||
|
||||
async def get_project_id(
|
||||
project_repository: ProjectRepositoryDep,
|
||||
project: ProjectPathDep,
|
||||
@@ -188,6 +248,17 @@ async def get_entity_repository(
|
||||
EntityRepositoryDep = Annotated[EntityRepository, Depends(get_entity_repository)]
|
||||
|
||||
|
||||
async def get_entity_repository_v2(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdPathDep,
|
||||
) -> EntityRepository:
|
||||
"""Create an EntityRepository instance for v2 API (uses integer project_id from path)."""
|
||||
return EntityRepository(session_maker, project_id=project_id)
|
||||
|
||||
|
||||
EntityRepositoryV2Dep = Annotated[EntityRepository, Depends(get_entity_repository_v2)]
|
||||
|
||||
|
||||
async def get_observation_repository(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdDep,
|
||||
@@ -199,6 +270,19 @@ async def get_observation_repository(
|
||||
ObservationRepositoryDep = Annotated[ObservationRepository, Depends(get_observation_repository)]
|
||||
|
||||
|
||||
async def get_observation_repository_v2(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdPathDep,
|
||||
) -> ObservationRepository:
|
||||
"""Create an ObservationRepository instance for v2 API."""
|
||||
return ObservationRepository(session_maker, project_id=project_id)
|
||||
|
||||
|
||||
ObservationRepositoryV2Dep = Annotated[
|
||||
ObservationRepository, Depends(get_observation_repository_v2)
|
||||
]
|
||||
|
||||
|
||||
async def get_relation_repository(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdDep,
|
||||
@@ -210,17 +294,43 @@ async def get_relation_repository(
|
||||
RelationRepositoryDep = Annotated[RelationRepository, Depends(get_relation_repository)]
|
||||
|
||||
|
||||
async def get_relation_repository_v2(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdPathDep,
|
||||
) -> RelationRepository:
|
||||
"""Create a RelationRepository instance for v2 API."""
|
||||
return RelationRepository(session_maker, project_id=project_id)
|
||||
|
||||
|
||||
RelationRepositoryV2Dep = Annotated[RelationRepository, Depends(get_relation_repository_v2)]
|
||||
|
||||
|
||||
async def get_search_repository(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdDep,
|
||||
) -> SearchRepository:
|
||||
"""Create a SearchRepository instance for the current project."""
|
||||
return SearchRepository(session_maker, project_id=project_id)
|
||||
"""Create a backend-specific SearchRepository instance for the current project.
|
||||
|
||||
Uses factory function to return SQLiteSearchRepository or PostgresSearchRepository
|
||||
based on database backend configuration.
|
||||
"""
|
||||
return create_search_repository(session_maker, project_id=project_id)
|
||||
|
||||
|
||||
SearchRepositoryDep = Annotated[SearchRepository, Depends(get_search_repository)]
|
||||
|
||||
|
||||
async def get_search_repository_v2(
|
||||
session_maker: SessionMakerDep,
|
||||
project_id: ProjectIdPathDep,
|
||||
) -> SearchRepository:
|
||||
"""Create a SearchRepository instance for v2 API."""
|
||||
return create_search_repository(session_maker, project_id=project_id)
|
||||
|
||||
|
||||
SearchRepositoryV2Dep = Annotated[SearchRepository, Depends(get_search_repository_v2)]
|
||||
|
||||
|
||||
# ProjectInfoRepository is deprecated and will be removed in a future version.
|
||||
# Use ProjectRepository instead, which has the same functionality plus more project-specific operations.
|
||||
|
||||
@@ -234,6 +344,13 @@ async def get_entity_parser(project_config: ProjectConfigDep) -> EntityParser:
|
||||
EntityParserDep = Annotated["EntityParser", Depends(get_entity_parser)]
|
||||
|
||||
|
||||
async def get_entity_parser_v2(project_config: ProjectConfigV2Dep) -> EntityParser:
|
||||
return EntityParser(project_config.home)
|
||||
|
||||
|
||||
EntityParserV2Dep = Annotated["EntityParser", Depends(get_entity_parser_v2)]
|
||||
|
||||
|
||||
async def get_markdown_processor(entity_parser: EntityParserDep) -> MarkdownProcessor:
|
||||
return MarkdownProcessor(entity_parser)
|
||||
|
||||
@@ -241,20 +358,39 @@ async def get_markdown_processor(entity_parser: EntityParserDep) -> MarkdownProc
|
||||
MarkdownProcessorDep = Annotated[MarkdownProcessor, Depends(get_markdown_processor)]
|
||||
|
||||
|
||||
async def get_markdown_processor_v2(entity_parser: EntityParserV2Dep) -> MarkdownProcessor:
|
||||
return MarkdownProcessor(entity_parser)
|
||||
|
||||
|
||||
MarkdownProcessorV2Dep = Annotated[MarkdownProcessor, Depends(get_markdown_processor_v2)]
|
||||
|
||||
|
||||
async def get_file_service(
|
||||
project_config: ProjectConfigDep, markdown_processor: MarkdownProcessorDep
|
||||
) -> FileService:
|
||||
logger.debug(
|
||||
f"Creating FileService for project: {project_config.name}, base_path: {project_config.home}"
|
||||
)
|
||||
file_service = FileService(project_config.home, markdown_processor)
|
||||
logger.debug(f"Created FileService for project: {file_service} ")
|
||||
logger.debug(
|
||||
f"Created FileService for project: {project_config.name}, base_path: {project_config.home} "
|
||||
)
|
||||
return file_service
|
||||
|
||||
|
||||
FileServiceDep = Annotated[FileService, Depends(get_file_service)]
|
||||
|
||||
|
||||
async def get_file_service_v2(
|
||||
project_config: ProjectConfigV2Dep, markdown_processor: MarkdownProcessorV2Dep
|
||||
) -> FileService:
|
||||
file_service = FileService(project_config.home, markdown_processor)
|
||||
logger.debug(
|
||||
f"Created FileService for project: {project_config.name}, base_path: {project_config.home}"
|
||||
)
|
||||
return file_service
|
||||
|
||||
|
||||
FileServiceV2Dep = Annotated[FileService, Depends(get_file_service_v2)]
|
||||
|
||||
|
||||
async def get_entity_service(
|
||||
entity_repository: EntityRepositoryDep,
|
||||
observation_repository: ObservationRepositoryDep,
|
||||
@@ -279,6 +415,30 @@ async def get_entity_service(
|
||||
EntityServiceDep = Annotated[EntityService, Depends(get_entity_service)]
|
||||
|
||||
|
||||
async def get_entity_service_v2(
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
observation_repository: ObservationRepositoryV2Dep,
|
||||
relation_repository: RelationRepositoryV2Dep,
|
||||
entity_parser: EntityParserV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
link_resolver: "LinkResolverV2Dep",
|
||||
app_config: AppConfigDep,
|
||||
) -> EntityService:
|
||||
"""Create EntityService for v2 API."""
|
||||
return EntityService(
|
||||
entity_repository=entity_repository,
|
||||
observation_repository=observation_repository,
|
||||
relation_repository=relation_repository,
|
||||
entity_parser=entity_parser,
|
||||
file_service=file_service,
|
||||
link_resolver=link_resolver,
|
||||
app_config=app_config,
|
||||
)
|
||||
|
||||
|
||||
EntityServiceV2Dep = Annotated[EntityService, Depends(get_entity_service_v2)]
|
||||
|
||||
|
||||
async def get_search_service(
|
||||
search_repository: SearchRepositoryDep,
|
||||
entity_repository: EntityRepositoryDep,
|
||||
@@ -291,6 +451,18 @@ async def get_search_service(
|
||||
SearchServiceDep = Annotated[SearchService, Depends(get_search_service)]
|
||||
|
||||
|
||||
async def get_search_service_v2(
|
||||
search_repository: SearchRepositoryV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
) -> SearchService:
|
||||
"""Create SearchService for v2 API."""
|
||||
return SearchService(search_repository, entity_repository, file_service)
|
||||
|
||||
|
||||
SearchServiceV2Dep = Annotated[SearchService, Depends(get_search_service_v2)]
|
||||
|
||||
|
||||
async def get_link_resolver(
|
||||
entity_repository: EntityRepositoryDep, search_service: SearchServiceDep
|
||||
) -> LinkResolver:
|
||||
@@ -300,6 +472,15 @@ async def get_link_resolver(
|
||||
LinkResolverDep = Annotated[LinkResolver, Depends(get_link_resolver)]
|
||||
|
||||
|
||||
async def get_link_resolver_v2(
|
||||
entity_repository: EntityRepositoryV2Dep, search_service: SearchServiceV2Dep
|
||||
) -> LinkResolver:
|
||||
return LinkResolver(entity_repository=entity_repository, search_service=search_service)
|
||||
|
||||
|
||||
LinkResolverV2Dep = Annotated[LinkResolver, Depends(get_link_resolver_v2)]
|
||||
|
||||
|
||||
async def get_context_service(
|
||||
search_repository: SearchRepositoryDep,
|
||||
entity_repository: EntityRepositoryDep,
|
||||
@@ -315,6 +496,22 @@ async def get_context_service(
|
||||
ContextServiceDep = Annotated[ContextService, Depends(get_context_service)]
|
||||
|
||||
|
||||
async def get_context_service_v2(
|
||||
search_repository: SearchRepositoryV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
observation_repository: ObservationRepositoryV2Dep,
|
||||
) -> ContextService:
|
||||
"""Create ContextService for v2 API."""
|
||||
return ContextService(
|
||||
search_repository=search_repository,
|
||||
entity_repository=entity_repository,
|
||||
observation_repository=observation_repository,
|
||||
)
|
||||
|
||||
|
||||
ContextServiceV2Dep = Annotated[ContextService, Depends(get_context_service_v2)]
|
||||
|
||||
|
||||
async def get_sync_service(
|
||||
app_config: AppConfigDep,
|
||||
entity_service: EntityServiceDep,
|
||||
@@ -344,6 +541,32 @@ async def get_sync_service(
|
||||
SyncServiceDep = Annotated[SyncService, Depends(get_sync_service)]
|
||||
|
||||
|
||||
async def get_sync_service_v2(
|
||||
app_config: AppConfigDep,
|
||||
entity_service: EntityServiceV2Dep,
|
||||
entity_parser: EntityParserV2Dep,
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
relation_repository: RelationRepositoryV2Dep,
|
||||
project_repository: ProjectRepositoryDep,
|
||||
search_service: SearchServiceV2Dep,
|
||||
file_service: FileServiceV2Dep,
|
||||
) -> SyncService: # pragma: no cover
|
||||
"""Create SyncService for v2 API."""
|
||||
return SyncService(
|
||||
app_config=app_config,
|
||||
entity_service=entity_service,
|
||||
entity_parser=entity_parser,
|
||||
entity_repository=entity_repository,
|
||||
relation_repository=relation_repository,
|
||||
project_repository=project_repository,
|
||||
search_service=search_service,
|
||||
file_service=file_service,
|
||||
)
|
||||
|
||||
|
||||
SyncServiceV2Dep = Annotated[SyncService, Depends(get_sync_service_v2)]
|
||||
|
||||
|
||||
async def get_project_service(
|
||||
project_repository: ProjectRepositoryDep,
|
||||
) -> ProjectService:
|
||||
@@ -366,6 +589,18 @@ async def get_directory_service(
|
||||
DirectoryServiceDep = Annotated[DirectoryService, Depends(get_directory_service)]
|
||||
|
||||
|
||||
async def get_directory_service_v2(
|
||||
entity_repository: EntityRepositoryV2Dep,
|
||||
) -> DirectoryService:
|
||||
"""Create DirectoryService for v2 API (uses integer project_id from path)."""
|
||||
return DirectoryService(
|
||||
entity_repository=entity_repository,
|
||||
)
|
||||
|
||||
|
||||
DirectoryServiceV2Dep = Annotated[DirectoryService, Depends(get_directory_service_v2)]
|
||||
|
||||
|
||||
# Import
|
||||
|
||||
|
||||
@@ -409,3 +644,50 @@ async def get_memory_json_importer(
|
||||
|
||||
|
||||
MemoryJsonImporterDep = Annotated[MemoryJsonImporter, Depends(get_memory_json_importer)]
|
||||
|
||||
|
||||
# V2 Import dependencies
|
||||
|
||||
|
||||
async def get_chatgpt_importer_v2(
|
||||
project_config: ProjectConfigV2Dep, markdown_processor: MarkdownProcessorV2Dep
|
||||
) -> ChatGPTImporter:
|
||||
"""Create ChatGPTImporter with v2 dependencies."""
|
||||
return ChatGPTImporter(project_config.home, markdown_processor)
|
||||
|
||||
|
||||
ChatGPTImporterV2Dep = Annotated[ChatGPTImporter, Depends(get_chatgpt_importer_v2)]
|
||||
|
||||
|
||||
async def get_claude_conversations_importer_v2(
|
||||
project_config: ProjectConfigV2Dep, markdown_processor: MarkdownProcessorV2Dep
|
||||
) -> ClaudeConversationsImporter:
|
||||
"""Create ClaudeConversationsImporter with v2 dependencies."""
|
||||
return ClaudeConversationsImporter(project_config.home, markdown_processor)
|
||||
|
||||
|
||||
ClaudeConversationsImporterV2Dep = Annotated[
|
||||
ClaudeConversationsImporter, Depends(get_claude_conversations_importer_v2)
|
||||
]
|
||||
|
||||
|
||||
async def get_claude_projects_importer_v2(
|
||||
project_config: ProjectConfigV2Dep, markdown_processor: MarkdownProcessorV2Dep
|
||||
) -> ClaudeProjectsImporter:
|
||||
"""Create ClaudeProjectsImporter with v2 dependencies."""
|
||||
return ClaudeProjectsImporter(project_config.home, markdown_processor)
|
||||
|
||||
|
||||
ClaudeProjectsImporterV2Dep = Annotated[
|
||||
ClaudeProjectsImporter, Depends(get_claude_projects_importer_v2)
|
||||
]
|
||||
|
||||
|
||||
async def get_memory_json_importer_v2(
|
||||
project_config: ProjectConfigV2Dep, markdown_processor: MarkdownProcessorV2Dep
|
||||
) -> MemoryJsonImporter:
|
||||
"""Create MemoryJsonImporter with v2 dependencies."""
|
||||
return MemoryJsonImporter(project_config.home, markdown_processor)
|
||||
|
||||
|
||||
MemoryJsonImporterV2Dep = Annotated[MemoryJsonImporter, Depends(get_memory_json_importer_v2)]
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
"""Utilities for file operations."""
|
||||
|
||||
import hashlib
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
import re
|
||||
from typing import Any, Dict, Union
|
||||
@@ -13,6 +15,20 @@ from loguru import logger
|
||||
from basic_memory.utils import FilePath
|
||||
|
||||
|
||||
@dataclass
|
||||
class FileMetadata:
|
||||
"""File metadata for cloud-compatible file operations.
|
||||
|
||||
This dataclass provides a cloud-agnostic way to represent file metadata,
|
||||
enabling S3FileService to return metadata from head_object responses
|
||||
instead of mock stat_result with zeros.
|
||||
"""
|
||||
|
||||
size: int
|
||||
created_at: datetime
|
||||
modified_at: datetime
|
||||
|
||||
|
||||
class FileError(Exception):
|
||||
"""Base exception for file operations."""
|
||||
|
||||
|
||||
@@ -40,10 +40,13 @@ class ClaudeConversationsImporter(Importer[ChatImportResult]):
|
||||
chats_imported = 0
|
||||
|
||||
for chat in conversations:
|
||||
# Get name, providing default for unnamed conversations
|
||||
chat_name = chat.get("name") or f"Conversation {chat.get('uuid', 'untitled')}"
|
||||
|
||||
# Convert to entity
|
||||
entity = self._format_chat_content(
|
||||
base_path=folder_path,
|
||||
name=chat["name"],
|
||||
name=chat_name,
|
||||
messages=chat["chat_messages"],
|
||||
created_at=chat["created_at"],
|
||||
modified_at=chat["updated_at"],
|
||||
|
||||
@@ -23,6 +23,7 @@ from basic_memory.markdown.schemas import (
|
||||
)
|
||||
from basic_memory.utils import parse_tags
|
||||
|
||||
|
||||
md = MarkdownIt().use(observation_plugin).use(relation_plugin)
|
||||
|
||||
|
||||
@@ -189,35 +190,63 @@ class EntityParser:
|
||||
return self.base_path / path
|
||||
|
||||
async def parse_file_content(self, absolute_path, file_content):
|
||||
# Parse frontmatter with proper error handling for malformed YAML (issue #185)
|
||||
try:
|
||||
post = frontmatter.loads(file_content)
|
||||
except yaml.YAMLError as e:
|
||||
# Log the YAML parsing error with file context
|
||||
logger.warning(
|
||||
f"Failed to parse YAML frontmatter in {absolute_path}: {e}. "
|
||||
f"Treating file as plain markdown without frontmatter."
|
||||
)
|
||||
# Create a post with no frontmatter - treat entire content as markdown
|
||||
post = frontmatter.Post(file_content, metadata={})
|
||||
"""Parse markdown content from file stats.
|
||||
|
||||
# Extract file stat info
|
||||
Delegates to parse_markdown_content() for actual parsing logic.
|
||||
Exists for backwards compatibility with code that passes file paths.
|
||||
"""
|
||||
# Extract file stat info for timestamps
|
||||
file_stats = absolute_path.stat()
|
||||
|
||||
# Normalize frontmatter values to prevent AttributeError on date objects (issue #236)
|
||||
# PyYAML automatically converts date strings like "2025-10-24" to datetime.date objects
|
||||
# This normalization converts them back to ISO format strings to ensure compatibility
|
||||
# with code that expects string values
|
||||
# Delegate to parse_markdown_content with timestamps from file stats
|
||||
return await self.parse_markdown_content(
|
||||
file_path=absolute_path,
|
||||
content=file_content,
|
||||
mtime=file_stats.st_mtime,
|
||||
ctime=file_stats.st_ctime,
|
||||
)
|
||||
|
||||
async def parse_markdown_content(
|
||||
self,
|
||||
file_path: Path,
|
||||
content: str,
|
||||
mtime: Optional[float] = None,
|
||||
ctime: Optional[float] = None,
|
||||
) -> EntityMarkdown:
|
||||
"""Parse markdown content without requiring file to exist on disk.
|
||||
|
||||
Useful for parsing content from S3 or other remote sources where the file
|
||||
is not available locally.
|
||||
|
||||
Args:
|
||||
file_path: Path for metadata (doesn't need to exist on disk)
|
||||
content: Markdown content as string
|
||||
mtime: Optional modification time (Unix timestamp)
|
||||
ctime: Optional creation time (Unix timestamp)
|
||||
|
||||
Returns:
|
||||
EntityMarkdown with parsed content
|
||||
"""
|
||||
# Parse frontmatter with proper error handling for malformed YAML
|
||||
try:
|
||||
post = frontmatter.loads(content)
|
||||
except yaml.YAMLError as e:
|
||||
logger.warning(
|
||||
f"Failed to parse YAML frontmatter in {file_path}: {e}. "
|
||||
f"Treating file as plain markdown without frontmatter."
|
||||
)
|
||||
post = frontmatter.Post(content, metadata={})
|
||||
|
||||
# Normalize frontmatter values
|
||||
metadata = normalize_frontmatter_metadata(post.metadata)
|
||||
|
||||
# Ensure required fields have defaults (issue #184, #387)
|
||||
# Handle title - use default if missing, None/null, empty, or string "None"
|
||||
# Ensure required fields have defaults
|
||||
title = metadata.get("title")
|
||||
if not title or title == "None":
|
||||
metadata["title"] = absolute_path.stem
|
||||
metadata["title"] = file_path.stem
|
||||
else:
|
||||
metadata["title"] = title
|
||||
# Handle type - use default if missing OR explicitly set to None/null
|
||||
|
||||
entity_type = metadata.get("type")
|
||||
metadata["type"] = entity_type if entity_type is not None else "note"
|
||||
|
||||
@@ -225,16 +254,20 @@ class EntityParser:
|
||||
if tags:
|
||||
metadata["tags"] = tags
|
||||
|
||||
# frontmatter - use metadata with defaults applied
|
||||
entity_frontmatter = EntityFrontmatter(
|
||||
metadata=metadata,
|
||||
)
|
||||
# Parse content for observations and relations
|
||||
entity_frontmatter = EntityFrontmatter(metadata=metadata)
|
||||
entity_content = parse(post.content)
|
||||
|
||||
# Use provided timestamps or current time as fallback
|
||||
now = datetime.now().astimezone()
|
||||
created = datetime.fromtimestamp(ctime).astimezone() if ctime else now
|
||||
modified = datetime.fromtimestamp(mtime).astimezone() if mtime else now
|
||||
|
||||
return EntityMarkdown(
|
||||
frontmatter=entity_frontmatter,
|
||||
content=post.content,
|
||||
observations=entity_content.observations,
|
||||
relations=entity_content.relations,
|
||||
created=datetime.fromtimestamp(file_stats.st_ctime).astimezone(),
|
||||
modified=datetime.fromtimestamp(file_stats.st_mtime).astimezone(),
|
||||
created=created,
|
||||
modified=modified,
|
||||
)
|
||||
|
||||
@@ -5,6 +5,7 @@ from collections import OrderedDict
|
||||
from frontmatter import Post
|
||||
from loguru import logger
|
||||
|
||||
|
||||
from basic_memory import file_utils
|
||||
from basic_memory.file_utils import dump_frontmatter
|
||||
from basic_memory.markdown.entity_parser import EntityParser
|
||||
|
||||
@@ -30,7 +30,9 @@ def is_observation(token: Token) -> bool:
|
||||
|
||||
# Check for proper observation format: [category] content
|
||||
match = re.match(r"^\[([^\[\]()]+)\]\s+(.+)", content)
|
||||
has_tags = "#" in content
|
||||
# Check for standalone hashtags (words starting with #)
|
||||
# This excludes # in HTML attributes like color="#4285F4"
|
||||
has_tags = any(part.startswith("#") for part in content.split())
|
||||
return bool(match) or has_tags
|
||||
|
||||
|
||||
@@ -160,7 +162,7 @@ def parse_inline_relations(content: str) -> List[Dict[str, Any]]:
|
||||
|
||||
target = content[start + 2 : end].strip()
|
||||
if target:
|
||||
relations.append({"type": "links to", "target": target, "context": None})
|
||||
relations.append({"type": "links_to", "target": target, "context": None})
|
||||
|
||||
start = end + 2
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
|
||||
from frontmatter import Post
|
||||
|
||||
from basic_memory.file_utils import has_frontmatter, remove_frontmatter, parse_frontmatter
|
||||
@@ -12,7 +13,10 @@ from basic_memory.models import Observation as ObservationModel
|
||||
|
||||
|
||||
def entity_model_from_markdown(
|
||||
file_path: Path, markdown: EntityMarkdown, entity: Optional[Entity] = None
|
||||
file_path: Path,
|
||||
markdown: EntityMarkdown,
|
||||
entity: Optional[Entity] = None,
|
||||
project_id: Optional[int] = None,
|
||||
) -> Entity:
|
||||
"""
|
||||
Convert markdown entity to model. Does not include relations.
|
||||
@@ -21,6 +25,7 @@ def entity_model_from_markdown(
|
||||
file_path: Path to the markdown file
|
||||
markdown: Parsed markdown entity
|
||||
entity: Optional existing entity to update
|
||||
project_id: Project ID for new observations (uses entity.project_id if not provided)
|
||||
|
||||
Returns:
|
||||
Entity model populated from markdown
|
||||
@@ -50,9 +55,13 @@ def entity_model_from_markdown(
|
||||
metadata = markdown.frontmatter.metadata or {}
|
||||
model.entity_metadata = {k: str(v) for k, v in metadata.items() if v is not None}
|
||||
|
||||
# Get project_id from entity if not provided
|
||||
obs_project_id = project_id or (model.project_id if hasattr(model, "project_id") else None)
|
||||
|
||||
# Convert observations
|
||||
model.observations = [
|
||||
ObservationModel(
|
||||
project_id=obs_project_id,
|
||||
content=obs.content,
|
||||
category=obs.category,
|
||||
context=obs.context,
|
||||
|
||||
@@ -129,7 +129,7 @@ class Entity(Base):
|
||||
return value
|
||||
|
||||
def __repr__(self) -> str:
|
||||
return f"Entity(id={self.id}, name='{self.title}', type='{self.entity_type}'"
|
||||
return f"Entity(id={self.id}, name='{self.title}', type='{self.entity_type}', checksum='{self.checksum}')"
|
||||
|
||||
|
||||
class Observation(Base):
|
||||
@@ -145,6 +145,7 @@ class Observation(Base):
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
project_id: Mapped[int] = mapped_column(Integer, ForeignKey("project.id"), index=True)
|
||||
entity_id: Mapped[int] = mapped_column(Integer, ForeignKey("entity.id", ondelete="CASCADE"))
|
||||
content: Mapped[str] = mapped_column(Text)
|
||||
category: Mapped[str] = mapped_column(String, nullable=False, default="note")
|
||||
@@ -162,9 +163,14 @@ class Observation(Base):
|
||||
|
||||
We can construct these because observations are always defined in
|
||||
and owned by a single entity.
|
||||
|
||||
Content is truncated to 200 chars to stay under PostgreSQL's
|
||||
btree index limit of 2704 bytes.
|
||||
"""
|
||||
# Truncate content to avoid exceeding PostgreSQL's btree index limit
|
||||
content_for_permalink = self.content[:200] if len(self.content) > 200 else self.content
|
||||
return generate_permalink(
|
||||
f"{self.entity.permalink}/observations/{self.category}/{self.content}"
|
||||
f"{self.entity.permalink}/observations/{self.category}/{content_for_permalink}"
|
||||
)
|
||||
|
||||
def __repr__(self) -> str: # pragma: no cover
|
||||
@@ -186,6 +192,7 @@ class Relation(Base):
|
||||
)
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True)
|
||||
project_id: Mapped[int] = mapped_column(Integer, ForeignKey("project.id"), index=True)
|
||||
from_id: Mapped[int] = mapped_column(Integer, ForeignKey("entity.id", ondelete="CASCADE"))
|
||||
to_id: Mapped[Optional[int]] = mapped_column(
|
||||
Integer, ForeignKey("entity.id", ondelete="CASCADE"), nullable=True
|
||||
|
||||
@@ -1,8 +1,55 @@
|
||||
"""Search models and tables."""
|
||||
"""Search DDL statements for SQLite and Postgres.
|
||||
|
||||
The search_index table is created via raw DDL, not ORM models, because:
|
||||
- SQLite uses FTS5 virtual tables (cannot be represented as ORM)
|
||||
- Postgres uses composite primary keys and generated tsvector columns
|
||||
- Both backends use raw SQL for all search operations via SearchIndexRow dataclass
|
||||
"""
|
||||
|
||||
from sqlalchemy import DDL
|
||||
|
||||
# Define FTS5 virtual table creation
|
||||
|
||||
# Define Postgres search_index table with composite primary key and tsvector
|
||||
# This DDL matches the Alembic migration schema (314f1ea54dc4)
|
||||
# Used by tests to create the table without running full migrations
|
||||
# NOTE: Split into separate DDL statements because asyncpg doesn't support
|
||||
# multiple statements in a single execute call.
|
||||
CREATE_POSTGRES_SEARCH_INDEX_TABLE = DDL("""
|
||||
CREATE TABLE IF NOT EXISTS search_index (
|
||||
id INTEGER NOT NULL,
|
||||
project_id INTEGER NOT NULL,
|
||||
title TEXT,
|
||||
content_stems TEXT,
|
||||
content_snippet TEXT,
|
||||
permalink VARCHAR,
|
||||
file_path VARCHAR,
|
||||
type VARCHAR,
|
||||
from_id INTEGER,
|
||||
to_id INTEGER,
|
||||
relation_type VARCHAR,
|
||||
entity_id INTEGER,
|
||||
category VARCHAR,
|
||||
metadata JSONB,
|
||||
created_at TIMESTAMP WITH TIME ZONE,
|
||||
updated_at TIMESTAMP WITH TIME ZONE,
|
||||
textsearchable_index_col tsvector GENERATED ALWAYS AS (
|
||||
to_tsvector('english', coalesce(title, '') || ' ' || coalesce(content_stems, ''))
|
||||
) STORED,
|
||||
PRIMARY KEY (id, type, project_id),
|
||||
FOREIGN KEY (project_id) REFERENCES project(id) ON DELETE CASCADE
|
||||
)
|
||||
""")
|
||||
|
||||
CREATE_POSTGRES_SEARCH_INDEX_FTS = DDL("""
|
||||
CREATE INDEX IF NOT EXISTS idx_search_index_fts ON search_index USING gin(textsearchable_index_col)
|
||||
""")
|
||||
|
||||
CREATE_POSTGRES_SEARCH_INDEX_METADATA = DDL("""
|
||||
CREATE INDEX IF NOT EXISTS idx_search_index_metadata_gin ON search_index USING gin(metadata jsonb_path_ops)
|
||||
""")
|
||||
|
||||
# Define FTS5 virtual table creation for SQLite only
|
||||
# This DDL is executed separately for SQLite databases
|
||||
CREATE_SEARCH_INDEX = DDL("""
|
||||
CREATE VIRTUAL TABLE IF NOT EXISTS search_index USING fts5(
|
||||
-- Core entity fields
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
"""Repository for managing entities in the knowledge graph."""
|
||||
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Sequence, Union
|
||||
from typing import List, Optional, Sequence, Union, Any
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import select
|
||||
@@ -9,6 +10,7 @@ from sqlalchemy.exc import IntegrityError
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
||||
from sqlalchemy.orm import selectinload
|
||||
from sqlalchemy.orm.interfaces import LoaderOption
|
||||
from sqlalchemy.engine import Row
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.models.knowledge import Entity, Observation, Relation
|
||||
@@ -31,6 +33,18 @@ class EntityRepository(Repository[Entity]):
|
||||
"""
|
||||
super().__init__(session_maker, Entity, project_id=project_id)
|
||||
|
||||
async def get_by_id(self, entity_id: int) -> Optional[Entity]:
|
||||
"""Get entity by numeric ID.
|
||||
|
||||
Args:
|
||||
entity_id: Numeric entity ID
|
||||
|
||||
Returns:
|
||||
Entity if found, None otherwise
|
||||
"""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
return await self.select_by_id(session, entity_id)
|
||||
|
||||
async def get_by_permalink(self, permalink: str) -> Optional[Entity]:
|
||||
"""Get entity by permalink.
|
||||
|
||||
@@ -63,6 +77,127 @@ class EntityRepository(Repository[Entity]):
|
||||
)
|
||||
return await self.find_one(query)
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Lightweight methods for permalink resolution (no eager loading)
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
async def permalink_exists(self, permalink: str) -> bool:
|
||||
"""Check if a permalink exists without loading the full entity.
|
||||
|
||||
This is much faster than get_by_permalink() as it skips eager loading
|
||||
of observations and relations. Use for existence checks in bulk operations.
|
||||
|
||||
Args:
|
||||
permalink: Permalink to check
|
||||
|
||||
Returns:
|
||||
True if permalink exists, False otherwise
|
||||
"""
|
||||
query = select(Entity.id).where(Entity.permalink == permalink).limit(1)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return result.scalar_one_or_none() is not None
|
||||
|
||||
async def get_file_path_for_permalink(self, permalink: str) -> Optional[str]:
|
||||
"""Get the file_path for a permalink without loading the full entity.
|
||||
|
||||
Use when you only need the file_path, not the full entity with relations.
|
||||
|
||||
Args:
|
||||
permalink: Permalink to look up
|
||||
|
||||
Returns:
|
||||
file_path string if found, None otherwise
|
||||
"""
|
||||
query = select(Entity.file_path).where(Entity.permalink == permalink)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return result.scalar_one_or_none()
|
||||
|
||||
async def get_permalink_for_file_path(self, file_path: Union[Path, str]) -> Optional[str]:
|
||||
"""Get the permalink for a file_path without loading the full entity.
|
||||
|
||||
Use when you only need the permalink, not the full entity with relations.
|
||||
|
||||
Args:
|
||||
file_path: File path to look up
|
||||
|
||||
Returns:
|
||||
permalink string if found, None otherwise
|
||||
"""
|
||||
query = select(Entity.permalink).where(Entity.file_path == Path(file_path).as_posix())
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return result.scalar_one_or_none()
|
||||
|
||||
async def get_all_permalinks(self) -> List[str]:
|
||||
"""Get all permalinks for this project.
|
||||
|
||||
Optimized for bulk operations - returns only permalink strings
|
||||
without loading entities or relationships.
|
||||
|
||||
Returns:
|
||||
List of all permalinks in the project
|
||||
"""
|
||||
query = select(Entity.permalink)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return list(result.scalars().all())
|
||||
|
||||
async def get_permalink_to_file_path_map(self) -> dict[str, str]:
|
||||
"""Get a mapping of permalink -> file_path for all entities.
|
||||
|
||||
Optimized for bulk permalink resolution - loads minimal data in one query.
|
||||
|
||||
Returns:
|
||||
Dict mapping permalink to file_path
|
||||
"""
|
||||
query = select(Entity.permalink, Entity.file_path)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return {row.permalink: row.file_path for row in result.all()}
|
||||
|
||||
async def get_file_path_to_permalink_map(self) -> dict[str, str]:
|
||||
"""Get a mapping of file_path -> permalink for all entities.
|
||||
|
||||
Optimized for bulk permalink resolution - loads minimal data in one query.
|
||||
|
||||
Returns:
|
||||
Dict mapping file_path to permalink
|
||||
"""
|
||||
query = select(Entity.file_path, Entity.permalink)
|
||||
query = self._add_project_filter(query)
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return {row.file_path: row.permalink for row in result.all()}
|
||||
|
||||
async def get_by_file_paths(
|
||||
self, session: AsyncSession, file_paths: Sequence[Union[Path, str]]
|
||||
) -> List[Row[Any]]:
|
||||
"""Get file paths and checksums for multiple entities (optimized for change detection).
|
||||
|
||||
Only queries file_path and checksum columns, skips loading full entities and relationships.
|
||||
This is much faster than loading complete Entity objects when you only need checksums.
|
||||
|
||||
Args:
|
||||
session: Database session to use for the query
|
||||
file_paths: List of file paths to query
|
||||
|
||||
Returns:
|
||||
List of (file_path, checksum) tuples for matching entities
|
||||
"""
|
||||
if not file_paths:
|
||||
return []
|
||||
|
||||
# Convert all paths to POSIX strings for consistent comparison
|
||||
posix_paths = [Path(fp).as_posix() for fp in file_paths]
|
||||
|
||||
# Query ONLY file_path and checksum columns (not full Entity objects)
|
||||
query = select(Entity.file_path, Entity.checksum).where(Entity.file_path.in_(posix_paths))
|
||||
query = self._add_project_filter(query)
|
||||
|
||||
result = await session.execute(query)
|
||||
return list(result.all())
|
||||
|
||||
async def find_by_checksum(self, checksum: str) -> Sequence[Entity]:
|
||||
"""Find entities with the given checksum.
|
||||
|
||||
@@ -80,6 +215,34 @@ class EntityRepository(Repository[Entity]):
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return list(result.scalars().all())
|
||||
|
||||
async def find_by_checksums(self, checksums: Sequence[str]) -> Sequence[Entity]:
|
||||
"""Find entities with any of the given checksums (batch query for move detection).
|
||||
|
||||
This is a batch-optimized version of find_by_checksum() that queries multiple checksums
|
||||
in a single database query. Used for efficient move detection in cloud indexing.
|
||||
|
||||
Performance: For 1000 new files, this makes 1 query vs 1000 individual queries (~100x faster).
|
||||
|
||||
Example:
|
||||
When processing new files, we check if any are actually moved files by finding
|
||||
entities with matching checksums at different paths.
|
||||
|
||||
Args:
|
||||
checksums: List of file content checksums to search for
|
||||
|
||||
Returns:
|
||||
Sequence of entities with matching checksums (may be empty).
|
||||
Multiple entities may have the same checksum if files were copied.
|
||||
"""
|
||||
if not checksums:
|
||||
return []
|
||||
|
||||
# Query: SELECT * FROM entities WHERE checksum IN (checksum1, checksum2, ...)
|
||||
query = self.select().where(Entity.checksum.in_(checksums))
|
||||
# Don't load relationships for move detection - we only need file_path and checksum
|
||||
result = await self.execute_query(query, use_query_options=False)
|
||||
return list(result.scalars().all())
|
||||
|
||||
async def delete_by_file_path(self, file_path: Union[Path, str]) -> bool:
|
||||
"""Delete entity with the provided file_path.
|
||||
|
||||
@@ -155,8 +318,13 @@ class EntityRepository(Repository[Entity]):
|
||||
|
||||
except IntegrityError as e:
|
||||
# Check if this is a FOREIGN KEY constraint failure
|
||||
# SQLite: "FOREIGN KEY constraint failed"
|
||||
# Postgres: "violates foreign key constraint"
|
||||
error_str = str(e)
|
||||
if "FOREIGN KEY constraint failed" in error_str:
|
||||
if (
|
||||
"FOREIGN KEY constraint failed" in error_str
|
||||
or "violates foreign key constraint" in error_str
|
||||
):
|
||||
# Import locally to avoid circular dependency (repository -> services -> repository)
|
||||
from basic_memory.services.exceptions import SyncFatalError
|
||||
|
||||
@@ -310,5 +478,26 @@ class EntityRepository(Repository[Entity]):
|
||||
|
||||
# Insert with unique permalink
|
||||
session.add(entity)
|
||||
await session.flush()
|
||||
try:
|
||||
await session.flush()
|
||||
except IntegrityError as e:
|
||||
# Check if this is a FOREIGN KEY constraint failure
|
||||
# SQLite: "FOREIGN KEY constraint failed"
|
||||
# Postgres: "violates foreign key constraint"
|
||||
error_str = str(e)
|
||||
if (
|
||||
"FOREIGN KEY constraint failed" in error_str
|
||||
or "violates foreign key constraint" in error_str
|
||||
):
|
||||
# Import locally to avoid circular dependency (repository -> services -> repository)
|
||||
from basic_memory.services.exceptions import SyncFatalError
|
||||
|
||||
# Project doesn't exist in database - this is a fatal sync error
|
||||
raise SyncFatalError(
|
||||
f"Cannot sync file '{entity.file_path}': "
|
||||
f"project_id={entity.project_id} does not exist in database. "
|
||||
f"The project may have been deleted. This sync will be terminated."
|
||||
) from e
|
||||
# Re-raise if not a foreign key error
|
||||
raise
|
||||
return entity
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from typing import Dict, List, Sequence
|
||||
|
||||
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy.ext.asyncio import async_sessionmaker
|
||||
|
||||
|
||||
@@ -0,0 +1,379 @@
|
||||
"""PostgreSQL tsvector-based search repository implementation."""
|
||||
|
||||
import json
|
||||
import re
|
||||
from datetime import datetime
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
from basic_memory.repository.search_repository_base import SearchRepositoryBase
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
|
||||
|
||||
class PostgresSearchRepository(SearchRepositoryBase):
|
||||
"""PostgreSQL tsvector implementation of search repository.
|
||||
|
||||
Uses PostgreSQL's full-text search capabilities with:
|
||||
- tsvector for document representation
|
||||
- tsquery for query representation
|
||||
- GIN indexes for performance
|
||||
- ts_rank() function for relevance scoring
|
||||
- JSONB containment operators for metadata search
|
||||
"""
|
||||
|
||||
async def init_search_index(self):
|
||||
"""Create Postgres table with tsvector column and GIN indexes.
|
||||
|
||||
Note: This is handled by Alembic migrations. This method is a no-op
|
||||
for Postgres as the schema is created via migrations.
|
||||
"""
|
||||
logger.info("PostgreSQL search index initialization handled by migrations")
|
||||
# Table creation is done via Alembic migrations
|
||||
# This includes:
|
||||
# - CREATE TABLE search_index (...)
|
||||
# - ADD COLUMN textsearchable_index_col tsvector GENERATED ALWAYS AS (...)
|
||||
# - CREATE INDEX USING GIN on textsearchable_index_col
|
||||
# - CREATE INDEX USING GIN on metadata jsonb_path_ops
|
||||
pass
|
||||
|
||||
def _prepare_search_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a search term for tsquery format.
|
||||
|
||||
Args:
|
||||
term: The search term to prepare
|
||||
is_prefix: Whether to add prefix search capability (:* operator)
|
||||
|
||||
Returns:
|
||||
Formatted search term for tsquery
|
||||
|
||||
For Postgres:
|
||||
- Boolean operators are converted to tsquery format (&, |, !)
|
||||
- Prefix matching uses the :* operator
|
||||
- Terms are sanitized to prevent tsquery syntax errors
|
||||
"""
|
||||
# Check for explicit boolean operators
|
||||
boolean_operators = [" AND ", " OR ", " NOT "]
|
||||
if any(op in f" {term} " for op in boolean_operators):
|
||||
return self._prepare_boolean_query(term)
|
||||
|
||||
# For non-Boolean queries, prepare single term
|
||||
return self._prepare_single_term(term, is_prefix)
|
||||
|
||||
def _prepare_boolean_query(self, query: str) -> str:
|
||||
"""Convert Boolean query to tsquery format.
|
||||
|
||||
Args:
|
||||
query: A Boolean query like "coffee AND brewing" or "(pour OR french) AND press"
|
||||
|
||||
Returns:
|
||||
tsquery-formatted string with & (AND), | (OR), ! (NOT) operators
|
||||
|
||||
Examples:
|
||||
"coffee AND brewing" -> "coffee & brewing"
|
||||
"(pour OR french) AND press" -> "(pour | french) & press"
|
||||
"coffee NOT decaf" -> "coffee & !decaf"
|
||||
"""
|
||||
# Replace Boolean operators with tsquery operators
|
||||
# Keep parentheses for grouping
|
||||
result = query
|
||||
result = re.sub(r"\bAND\b", "&", result)
|
||||
result = re.sub(r"\bOR\b", "|", result)
|
||||
# NOT must be converted to "& !" and the ! must be attached to the following term
|
||||
# "Python NOT Django" -> "Python & !Django"
|
||||
result = re.sub(r"\bNOT\s+", "& !", result)
|
||||
|
||||
return result
|
||||
|
||||
def _prepare_single_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a single search term for tsquery.
|
||||
|
||||
Args:
|
||||
term: A single search term
|
||||
is_prefix: Whether to add prefix search capability (:* suffix)
|
||||
|
||||
Returns:
|
||||
A properly formatted single term for tsquery
|
||||
|
||||
For Postgres tsquery:
|
||||
- Multi-word queries become "word1 & word2"
|
||||
- Prefix matching uses ":*" suffix (e.g., "coff:*")
|
||||
- Special characters that need escaping: & | ! ( ) :
|
||||
"""
|
||||
if not term or not term.strip():
|
||||
return term
|
||||
|
||||
term = term.strip()
|
||||
|
||||
# Check if term is already a wildcard pattern
|
||||
if "*" in term:
|
||||
# Replace * with :* for Postgres prefix matching
|
||||
return term.replace("*", ":*")
|
||||
|
||||
# Remove tsquery special characters from the search term
|
||||
# These characters have special meaning in tsquery and cause syntax errors
|
||||
# if not used as operators
|
||||
special_chars = ["&", "|", "!", "(", ")", ":"]
|
||||
cleaned_term = term
|
||||
for char in special_chars:
|
||||
cleaned_term = cleaned_term.replace(char, " ")
|
||||
|
||||
# Handle multi-word queries
|
||||
if " " in cleaned_term:
|
||||
words = [w for w in cleaned_term.split() if w.strip()]
|
||||
if not words:
|
||||
# All characters were special chars, search won't match anything
|
||||
# Return a safe search term that won't cause syntax errors
|
||||
return "NOSPECIALCHARS:*"
|
||||
if is_prefix:
|
||||
# Add prefix matching to each word
|
||||
prepared_words = [f"{word}:*" for word in words]
|
||||
else:
|
||||
prepared_words = words
|
||||
# Join with AND operator
|
||||
return " & ".join(prepared_words)
|
||||
|
||||
# Single word
|
||||
cleaned_term = cleaned_term.strip()
|
||||
if not cleaned_term:
|
||||
return "NOSPECIALCHARS:*"
|
||||
if is_prefix:
|
||||
return f"{cleaned_term}:*"
|
||||
else:
|
||||
return cleaned_term
|
||||
|
||||
async def search(
|
||||
self,
|
||||
search_text: Optional[str] = None,
|
||||
permalink: Optional[str] = None,
|
||||
permalink_match: Optional[str] = None,
|
||||
title: Optional[str] = None,
|
||||
types: Optional[List[str]] = None,
|
||||
after_date: Optional[datetime] = None,
|
||||
search_item_types: Optional[List[SearchItemType]] = None,
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content using PostgreSQL tsvector."""
|
||||
conditions = []
|
||||
params = {}
|
||||
order_by_clause = ""
|
||||
|
||||
# Handle text search for title and content using tsvector
|
||||
if search_text:
|
||||
if search_text.strip() == "*" or search_text.strip() == "":
|
||||
# For wildcard searches, don't add any text conditions
|
||||
pass
|
||||
else:
|
||||
# Prepare search term for tsquery
|
||||
processed_text = self._prepare_search_term(search_text.strip())
|
||||
params["text"] = processed_text
|
||||
# Use @@ operator for tsvector matching
|
||||
conditions.append("textsearchable_index_col @@ to_tsquery('english', :text)")
|
||||
|
||||
# Handle title search
|
||||
if title:
|
||||
title_text = self._prepare_search_term(title.strip(), is_prefix=False)
|
||||
params["title_text"] = title_text
|
||||
conditions.append("to_tsvector('english', title) @@ to_tsquery('english', :title_text)")
|
||||
|
||||
# Handle permalink exact search
|
||||
if permalink:
|
||||
params["permalink"] = permalink
|
||||
conditions.append("permalink = :permalink")
|
||||
|
||||
# Handle permalink pattern match
|
||||
if permalink_match:
|
||||
permalink_text = permalink_match.lower().strip()
|
||||
params["permalink"] = permalink_text
|
||||
if "*" in permalink_match:
|
||||
# Use LIKE for pattern matching in Postgres
|
||||
# Convert * to % for SQL LIKE
|
||||
permalink_pattern = permalink_text.replace("*", "%")
|
||||
params["permalink"] = permalink_pattern
|
||||
conditions.append("permalink LIKE :permalink")
|
||||
else:
|
||||
conditions.append("permalink = :permalink")
|
||||
|
||||
# Handle search item type filter
|
||||
if search_item_types:
|
||||
type_list = ", ".join(f"'{t.value}'" for t in search_item_types)
|
||||
conditions.append(f"type IN ({type_list})")
|
||||
|
||||
# Handle entity type filter using JSONB containment
|
||||
if types:
|
||||
# Use JSONB @> operator for efficient containment queries
|
||||
type_conditions = []
|
||||
for entity_type in types:
|
||||
# Create JSONB containment condition for each type
|
||||
type_conditions.append(f'metadata @> \'{{"entity_type": "{entity_type}"}}\'')
|
||||
conditions.append(f"({' OR '.join(type_conditions)})")
|
||||
|
||||
# Handle date filter
|
||||
if after_date:
|
||||
params["after_date"] = after_date
|
||||
conditions.append("created_at > :after_date")
|
||||
# order by most recent first
|
||||
order_by_clause = ", updated_at DESC"
|
||||
|
||||
# Always filter by project_id
|
||||
params["project_id"] = self.project_id
|
||||
conditions.append("project_id = :project_id")
|
||||
|
||||
# set limit and offset
|
||||
params["limit"] = limit
|
||||
params["offset"] = offset
|
||||
|
||||
# Build WHERE clause
|
||||
where_clause = " AND ".join(conditions) if conditions else "1=1"
|
||||
|
||||
# Build SQL with ts_rank() for scoring
|
||||
# Note: If no text search, score will be NULL, so we use COALESCE to default to 0
|
||||
if search_text and search_text.strip() and search_text.strip() != "*":
|
||||
score_expr = "ts_rank(textsearchable_index_col, to_tsquery('english', :text))"
|
||||
else:
|
||||
score_expr = "0"
|
||||
|
||||
sql = f"""
|
||||
SELECT
|
||||
project_id,
|
||||
id,
|
||||
title,
|
||||
permalink,
|
||||
file_path,
|
||||
type,
|
||||
metadata,
|
||||
from_id,
|
||||
to_id,
|
||||
relation_type,
|
||||
entity_id,
|
||||
content_snippet,
|
||||
category,
|
||||
created_at,
|
||||
updated_at,
|
||||
{score_expr} as score
|
||||
FROM search_index
|
||||
WHERE {where_clause}
|
||||
ORDER BY score DESC, id ASC {order_by_clause}
|
||||
LIMIT :limit
|
||||
OFFSET :offset
|
||||
"""
|
||||
|
||||
logger.trace(f"Search {sql} params: {params}")
|
||||
try:
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
except Exception as e:
|
||||
# Handle tsquery syntax errors
|
||||
if "tsquery" in str(e).lower() or "syntax error" in str(e).lower(): # pragma: no cover
|
||||
logger.warning(f"tsquery syntax error for search term: {search_text}, error: {e}")
|
||||
# Return empty results rather than crashing
|
||||
return []
|
||||
else:
|
||||
# Re-raise other database errors
|
||||
logger.error(f"Database error during search: {e}")
|
||||
raise
|
||||
|
||||
results = [
|
||||
SearchIndexRow(
|
||||
project_id=self.project_id,
|
||||
id=row.id,
|
||||
title=row.title,
|
||||
permalink=row.permalink,
|
||||
file_path=row.file_path,
|
||||
type=row.type,
|
||||
score=float(row.score) if row.score else 0.0,
|
||||
metadata=(
|
||||
row.metadata
|
||||
if isinstance(row.metadata, dict)
|
||||
else (json.loads(row.metadata) if row.metadata else {})
|
||||
),
|
||||
from_id=row.from_id,
|
||||
to_id=row.to_id,
|
||||
relation_type=row.relation_type,
|
||||
entity_id=row.entity_id,
|
||||
content_snippet=row.content_snippet,
|
||||
category=row.category,
|
||||
created_at=row.created_at,
|
||||
updated_at=row.updated_at,
|
||||
)
|
||||
for row in rows
|
||||
]
|
||||
|
||||
logger.trace(f"Found {len(results)} search results")
|
||||
for r in results:
|
||||
logger.trace(
|
||||
f"Search result: project_id: {r.project_id} type:{r.type} title: {r.title} permalink: {r.permalink} score: {r.score}"
|
||||
)
|
||||
|
||||
return results
|
||||
|
||||
async def bulk_index_items(self, search_index_rows: List[SearchIndexRow]) -> None:
|
||||
"""Index multiple items in a single batch operation using UPSERT.
|
||||
|
||||
Uses INSERT ... ON CONFLICT DO UPDATE to handle re-indexing of existing
|
||||
entities (e.g., during forward reference resolution) without requiring
|
||||
a separate delete operation. This eliminates race conditions between
|
||||
delete and insert operations in separate transactions.
|
||||
|
||||
Args:
|
||||
search_index_rows: List of SearchIndexRow objects to index
|
||||
"""
|
||||
|
||||
if not search_index_rows:
|
||||
return
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# When using text() raw SQL, always serialize JSON to string
|
||||
# Both SQLite (TEXT) and Postgres (JSONB) accept JSON strings in raw SQL
|
||||
# The database driver/column type will handle conversion
|
||||
insert_data_list = []
|
||||
for row in search_index_rows:
|
||||
insert_data = row.to_insert(serialize_json=True)
|
||||
insert_data["project_id"] = self.project_id
|
||||
insert_data_list.append(insert_data)
|
||||
|
||||
# Use UPSERT (INSERT ... ON CONFLICT) to handle re-indexing
|
||||
# Primary key is (id, type, project_id)
|
||||
# This handles race conditions during forward reference resolution
|
||||
# where an entity might be re-indexed before the delete commits
|
||||
# Syntax works for both SQLite 3.24+ and PostgreSQL
|
||||
await session.execute(
|
||||
text("""
|
||||
INSERT INTO search_index (
|
||||
id, title, content_stems, content_snippet, permalink, file_path, type, metadata,
|
||||
from_id, to_id, relation_type,
|
||||
entity_id, category,
|
||||
created_at, updated_at,
|
||||
project_id
|
||||
) VALUES (
|
||||
:id, :title, :content_stems, :content_snippet, :permalink, :file_path, :type, :metadata,
|
||||
:from_id, :to_id, :relation_type,
|
||||
:entity_id, :category,
|
||||
:created_at, :updated_at,
|
||||
:project_id
|
||||
)
|
||||
ON CONFLICT (id, type, project_id) DO UPDATE SET
|
||||
title = EXCLUDED.title,
|
||||
content_stems = EXCLUDED.content_stems,
|
||||
content_snippet = EXCLUDED.content_snippet,
|
||||
permalink = EXCLUDED.permalink,
|
||||
file_path = EXCLUDED.file_path,
|
||||
metadata = EXCLUDED.metadata,
|
||||
from_id = EXCLUDED.from_id,
|
||||
to_id = EXCLUDED.to_id,
|
||||
relation_type = EXCLUDED.relation_type,
|
||||
entity_id = EXCLUDED.entity_id,
|
||||
category = EXCLUDED.category,
|
||||
created_at = EXCLUDED.created_at,
|
||||
updated_at = EXCLUDED.updated_at
|
||||
"""),
|
||||
insert_data_list,
|
||||
)
|
||||
logger.debug(f"Bulk indexed {len(search_index_rows)} rows")
|
||||
await session.commit()
|
||||
@@ -3,6 +3,7 @@
|
||||
from pathlib import Path
|
||||
from typing import Optional, Sequence, Union
|
||||
|
||||
|
||||
from sqlalchemy import text
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
||||
|
||||
@@ -49,6 +50,18 @@ class ProjectRepository(Repository[Project]):
|
||||
query = self.select().where(Project.path == Path(path).as_posix())
|
||||
return await self.find_one(query)
|
||||
|
||||
async def get_by_id(self, project_id: int) -> Optional[Project]:
|
||||
"""Get project by numeric ID.
|
||||
|
||||
Args:
|
||||
project_id: Numeric project ID
|
||||
|
||||
Returns:
|
||||
Project if found, None otherwise
|
||||
"""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
return await self.select_by_id(session, project_id)
|
||||
|
||||
async def get_default_project(self) -> Optional[Project]:
|
||||
"""Get the default project (the one marked as is_default=True)."""
|
||||
query = self.select().where(Project.is_default.is_not(None))
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
"""Repository for managing Relation objects."""
|
||||
|
||||
from sqlalchemy import and_, delete
|
||||
from typing import Sequence, List, Optional
|
||||
|
||||
from sqlalchemy import select
|
||||
|
||||
from sqlalchemy import and_, delete, select
|
||||
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||
from sqlalchemy.ext.asyncio import async_sessionmaker
|
||||
from sqlalchemy.orm import selectinload, aliased
|
||||
from sqlalchemy.orm.interfaces import LoaderOption
|
||||
@@ -86,5 +88,59 @@ class RelationRepository(Repository[Relation]):
|
||||
result = await self.execute_query(query)
|
||||
return result.scalars().all()
|
||||
|
||||
async def add_all_ignore_duplicates(self, relations: List[Relation]) -> int:
|
||||
"""Bulk insert relations, ignoring duplicates.
|
||||
|
||||
Uses ON CONFLICT DO NOTHING to skip relations that would violate the
|
||||
unique constraint on (from_id, to_name, relation_type). This is useful
|
||||
for bulk operations where the same link may appear multiple times in
|
||||
a document.
|
||||
|
||||
Works with both SQLite and PostgreSQL dialects.
|
||||
|
||||
Args:
|
||||
relations: List of Relation objects to insert
|
||||
|
||||
Returns:
|
||||
Number of relations actually inserted (excludes duplicates)
|
||||
"""
|
||||
if not relations:
|
||||
return 0
|
||||
|
||||
# Convert Relation objects to dicts for insert
|
||||
values = [
|
||||
{
|
||||
"project_id": r.project_id if r.project_id else self.project_id,
|
||||
"from_id": r.from_id,
|
||||
"to_id": r.to_id,
|
||||
"to_name": r.to_name,
|
||||
"relation_type": r.relation_type,
|
||||
"context": r.context,
|
||||
}
|
||||
for r in relations
|
||||
]
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# Check dialect to use appropriate insert
|
||||
dialect_name = session.bind.dialect.name if session.bind else "sqlite"
|
||||
|
||||
if dialect_name == "postgresql":
|
||||
# PostgreSQL: use RETURNING to count inserted rows
|
||||
# (rowcount is 0 for ON CONFLICT DO NOTHING)
|
||||
stmt = (
|
||||
pg_insert(Relation)
|
||||
.values(values)
|
||||
.on_conflict_do_nothing()
|
||||
.returning(Relation.id)
|
||||
)
|
||||
result = await session.execute(stmt)
|
||||
return len(result.fetchall())
|
||||
else:
|
||||
# SQLite: rowcount works correctly
|
||||
stmt = sqlite_insert(Relation).values(values)
|
||||
stmt = stmt.on_conflict_do_nothing()
|
||||
result = await session.execute(stmt)
|
||||
return result.rowcount if result.rowcount > 0 else 0
|
||||
|
||||
def get_load_options(self) -> List[LoaderOption]:
|
||||
return [selectinload(Relation.from_entity), selectinload(Relation.to_entity)]
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from typing import Type, Optional, Any, Sequence, TypeVar, List, Dict
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import (
|
||||
select,
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
"""Search index data structures."""
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
from pathlib import Path
|
||||
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchIndexRow:
|
||||
"""Search result with score and metadata."""
|
||||
|
||||
project_id: int
|
||||
id: int
|
||||
type: str
|
||||
file_path: str
|
||||
|
||||
# date values
|
||||
created_at: datetime
|
||||
updated_at: datetime
|
||||
|
||||
permalink: Optional[str] = None
|
||||
metadata: Optional[dict] = None
|
||||
|
||||
# assigned in result
|
||||
score: Optional[float] = None
|
||||
|
||||
# Type-specific fields
|
||||
title: Optional[str] = None # entity
|
||||
content_stems: Optional[str] = None # entity, observation
|
||||
content_snippet: Optional[str] = None # entity, observation
|
||||
entity_id: Optional[int] = None # observations
|
||||
category: Optional[str] = None # observations
|
||||
from_id: Optional[int] = None # relations
|
||||
to_id: Optional[int] = None # relations
|
||||
relation_type: Optional[str] = None # relations
|
||||
|
||||
@property
|
||||
def content(self):
|
||||
return self.content_snippet
|
||||
|
||||
@property
|
||||
def directory(self) -> str:
|
||||
"""Extract directory part from file_path.
|
||||
|
||||
For a file at "projects/notes/ideas.md", returns "/projects/notes"
|
||||
For a file at root level "README.md", returns "/"
|
||||
"""
|
||||
if not self.type == SearchItemType.ENTITY.value and not self.file_path:
|
||||
return ""
|
||||
|
||||
# Normalize path separators to handle both Windows (\) and Unix (/) paths
|
||||
normalized_path = Path(self.file_path).as_posix()
|
||||
|
||||
# Split the path by slashes
|
||||
parts = normalized_path.split("/")
|
||||
|
||||
# If there's only one part (e.g., "README.md"), it's at the root
|
||||
if len(parts) <= 1:
|
||||
return "/"
|
||||
|
||||
# Join all parts except the last one (filename)
|
||||
directory_path = "/".join(parts[:-1])
|
||||
return f"/{directory_path}"
|
||||
|
||||
def to_insert(self, serialize_json: bool = True):
|
||||
"""Convert to dict for database insertion.
|
||||
|
||||
Args:
|
||||
serialize_json: If True, converts metadata dict to JSON string (for SQLite).
|
||||
If False, keeps metadata as dict (for Postgres JSONB).
|
||||
"""
|
||||
return {
|
||||
"id": self.id,
|
||||
"title": self.title,
|
||||
"content_stems": self.content_stems,
|
||||
"content_snippet": self.content_snippet,
|
||||
"permalink": self.permalink,
|
||||
"file_path": self.file_path,
|
||||
"type": self.type,
|
||||
"metadata": json.dumps(self.metadata)
|
||||
if serialize_json and self.metadata
|
||||
else self.metadata,
|
||||
"from_id": self.from_id,
|
||||
"to_id": self.to_id,
|
||||
"relation_type": self.relation_type,
|
||||
"entity_id": self.entity_id,
|
||||
"category": self.category,
|
||||
"created_at": self.created_at if self.created_at else None,
|
||||
"updated_at": self.updated_at if self.updated_at else None,
|
||||
"project_id": self.project_id,
|
||||
}
|
||||
@@ -1,365 +1,35 @@
|
||||
"""Repository for search operations."""
|
||||
"""Repository for search operations.
|
||||
|
||||
This module provides the search repository interface.
|
||||
The actual repository implementations are backend-specific:
|
||||
- SQLiteSearchRepository: Uses FTS5 virtual tables
|
||||
- PostgresSearchRepository: Uses tsvector/tsquery with GIN indexes
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Protocol
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import Executable, Result, text
|
||||
from sqlalchemy import Result
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.models.search import CREATE_SEARCH_INDEX
|
||||
from basic_memory.config import ConfigManager, DatabaseBackend
|
||||
from basic_memory.repository.postgres_search_repository import PostgresSearchRepository
|
||||
from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
from basic_memory.repository.sqlite_search_repository import SQLiteSearchRepository
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchIndexRow:
|
||||
"""Search result with score and metadata."""
|
||||
class SearchRepository(Protocol):
|
||||
"""Protocol defining the search repository interface.
|
||||
|
||||
Both SQLite and Postgres implementations must satisfy this protocol.
|
||||
"""
|
||||
|
||||
project_id: int
|
||||
id: int
|
||||
type: str
|
||||
file_path: str
|
||||
|
||||
# date values
|
||||
created_at: datetime
|
||||
updated_at: datetime
|
||||
|
||||
permalink: Optional[str] = None
|
||||
metadata: Optional[dict] = None
|
||||
|
||||
# assigned in result
|
||||
score: Optional[float] = None
|
||||
|
||||
# Type-specific fields
|
||||
title: Optional[str] = None # entity
|
||||
content_stems: Optional[str] = None # entity, observation
|
||||
content_snippet: Optional[str] = None # entity, observation
|
||||
entity_id: Optional[int] = None # observations
|
||||
category: Optional[str] = None # observations
|
||||
from_id: Optional[int] = None # relations
|
||||
to_id: Optional[int] = None # relations
|
||||
relation_type: Optional[str] = None # relations
|
||||
|
||||
@property
|
||||
def content(self):
|
||||
return self.content_snippet
|
||||
|
||||
@property
|
||||
def directory(self) -> str:
|
||||
"""Extract directory part from file_path.
|
||||
|
||||
For a file at "projects/notes/ideas.md", returns "/projects/notes"
|
||||
For a file at root level "README.md", returns "/"
|
||||
"""
|
||||
if not self.type == SearchItemType.ENTITY.value and not self.file_path:
|
||||
return ""
|
||||
|
||||
# Normalize path separators to handle both Windows (\) and Unix (/) paths
|
||||
normalized_path = Path(self.file_path).as_posix()
|
||||
|
||||
# Split the path by slashes
|
||||
parts = normalized_path.split("/")
|
||||
|
||||
# If there's only one part (e.g., "README.md"), it's at the root
|
||||
if len(parts) <= 1:
|
||||
return "/"
|
||||
|
||||
# Join all parts except the last one (filename)
|
||||
directory_path = "/".join(parts[:-1])
|
||||
return f"/{directory_path}"
|
||||
|
||||
def to_insert(self):
|
||||
return {
|
||||
"id": self.id,
|
||||
"title": self.title,
|
||||
"content_stems": self.content_stems,
|
||||
"content_snippet": self.content_snippet,
|
||||
"permalink": self.permalink,
|
||||
"file_path": self.file_path,
|
||||
"type": self.type,
|
||||
"metadata": json.dumps(self.metadata),
|
||||
"from_id": self.from_id,
|
||||
"to_id": self.to_id,
|
||||
"relation_type": self.relation_type,
|
||||
"entity_id": self.entity_id,
|
||||
"category": self.category,
|
||||
"created_at": self.created_at if self.created_at else None,
|
||||
"updated_at": self.updated_at if self.updated_at else None,
|
||||
"project_id": self.project_id,
|
||||
}
|
||||
|
||||
|
||||
class SearchRepository:
|
||||
"""Repository for search index operations."""
|
||||
|
||||
def __init__(self, session_maker: async_sessionmaker[AsyncSession], project_id: int):
|
||||
"""Initialize with session maker and project_id filter.
|
||||
|
||||
Args:
|
||||
session_maker: SQLAlchemy session maker
|
||||
project_id: Project ID to filter all operations by
|
||||
|
||||
Raises:
|
||||
ValueError: If project_id is None or invalid
|
||||
"""
|
||||
if project_id is None or project_id <= 0: # pragma: no cover
|
||||
raise ValueError("A valid project_id is required for SearchRepository")
|
||||
|
||||
self.session_maker = session_maker
|
||||
self.project_id = project_id
|
||||
|
||||
async def init_search_index(self):
|
||||
"""Create or recreate the search index."""
|
||||
logger.info("Initializing search index")
|
||||
try:
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
await session.execute(CREATE_SEARCH_INDEX)
|
||||
await session.commit()
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.error(f"Error initializing search index: {e}")
|
||||
raise e
|
||||
|
||||
def _prepare_boolean_query(self, query: str) -> str:
|
||||
"""Prepare a Boolean query by quoting individual terms while preserving operators.
|
||||
|
||||
Args:
|
||||
query: A Boolean query like "tier1-test AND unicode" or "(hello OR world) NOT test"
|
||||
|
||||
Returns:
|
||||
A properly formatted Boolean query with quoted terms that need quoting
|
||||
"""
|
||||
# Define Boolean operators and their boundaries
|
||||
boolean_pattern = r"(\bAND\b|\bOR\b|\bNOT\b)"
|
||||
|
||||
# Split the query by Boolean operators, keeping the operators
|
||||
parts = re.split(boolean_pattern, query)
|
||||
|
||||
processed_parts = []
|
||||
for part in parts:
|
||||
part = part.strip()
|
||||
if not part:
|
||||
continue
|
||||
|
||||
# If it's a Boolean operator, keep it as is
|
||||
if part in ["AND", "OR", "NOT"]:
|
||||
processed_parts.append(part)
|
||||
else:
|
||||
# Handle parentheses specially - they should be preserved for grouping
|
||||
if "(" in part or ")" in part:
|
||||
# Parse parenthetical expressions carefully
|
||||
processed_part = self._prepare_parenthetical_term(part)
|
||||
processed_parts.append(processed_part)
|
||||
else:
|
||||
# This is a search term - for Boolean queries, don't add prefix wildcards
|
||||
prepared_term = self._prepare_single_term(part, is_prefix=False)
|
||||
processed_parts.append(prepared_term)
|
||||
|
||||
return " ".join(processed_parts)
|
||||
|
||||
def _prepare_parenthetical_term(self, term: str) -> str:
|
||||
"""Prepare a term that contains parentheses, preserving the parentheses for grouping.
|
||||
|
||||
Args:
|
||||
term: A term that may contain parentheses like "(hello" or "world)" or "(hello OR world)"
|
||||
|
||||
Returns:
|
||||
A properly formatted term with parentheses preserved
|
||||
"""
|
||||
# Handle terms that start/end with parentheses but may contain quotable content
|
||||
result = ""
|
||||
i = 0
|
||||
while i < len(term):
|
||||
if term[i] in "()":
|
||||
# Preserve parentheses as-is
|
||||
result += term[i]
|
||||
i += 1
|
||||
else:
|
||||
# Find the next parenthesis or end of string
|
||||
start = i
|
||||
while i < len(term) and term[i] not in "()":
|
||||
i += 1
|
||||
|
||||
# Extract the content between parentheses
|
||||
content = term[start:i].strip()
|
||||
if content:
|
||||
# Only quote if it actually needs quoting (has hyphens, special chars, etc)
|
||||
# but don't quote if it's just simple words
|
||||
if self._needs_quoting(content):
|
||||
escaped_content = content.replace('"', '""')
|
||||
result += f'"{escaped_content}"'
|
||||
else:
|
||||
result += content
|
||||
|
||||
return result
|
||||
|
||||
def _needs_quoting(self, term: str) -> bool:
|
||||
"""Check if a term needs to be quoted for FTS5 safety.
|
||||
|
||||
Args:
|
||||
term: The term to check
|
||||
|
||||
Returns:
|
||||
True if the term should be quoted
|
||||
"""
|
||||
if not term or not term.strip():
|
||||
return False
|
||||
|
||||
# Characters that indicate we should quote (excluding parentheses which are valid syntax)
|
||||
needs_quoting_chars = [
|
||||
" ",
|
||||
".",
|
||||
":",
|
||||
";",
|
||||
",",
|
||||
"<",
|
||||
">",
|
||||
"?",
|
||||
"/",
|
||||
"-",
|
||||
"'",
|
||||
'"',
|
||||
"[",
|
||||
"]",
|
||||
"{",
|
||||
"}",
|
||||
"+",
|
||||
"!",
|
||||
"@",
|
||||
"#",
|
||||
"$",
|
||||
"%",
|
||||
"^",
|
||||
"&",
|
||||
"=",
|
||||
"|",
|
||||
"\\",
|
||||
"~",
|
||||
"`",
|
||||
]
|
||||
|
||||
return any(c in term for c in needs_quoting_chars)
|
||||
|
||||
def _prepare_single_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a single search term (no Boolean operators).
|
||||
|
||||
Args:
|
||||
term: A single search term
|
||||
is_prefix: Whether to add prefix search capability (* suffix)
|
||||
|
||||
Returns:
|
||||
A properly formatted single term
|
||||
"""
|
||||
if not term or not term.strip():
|
||||
return term
|
||||
|
||||
term = term.strip()
|
||||
|
||||
# Check if term is already a proper wildcard pattern (alphanumeric + *)
|
||||
# e.g., "hello*", "test*world" - these should be left alone
|
||||
if "*" in term and all(c.isalnum() or c in "*_-" for c in term):
|
||||
return term
|
||||
|
||||
# Characters that can cause FTS5 syntax errors when used as operators
|
||||
# We're more conservative here - only quote when we detect problematic patterns
|
||||
problematic_chars = [
|
||||
'"',
|
||||
"'",
|
||||
"(",
|
||||
")",
|
||||
"[",
|
||||
"]",
|
||||
"{",
|
||||
"}",
|
||||
"+",
|
||||
"!",
|
||||
"@",
|
||||
"#",
|
||||
"$",
|
||||
"%",
|
||||
"^",
|
||||
"&",
|
||||
"=",
|
||||
"|",
|
||||
"\\",
|
||||
"~",
|
||||
"`",
|
||||
]
|
||||
|
||||
# Characters that indicate we should quote (spaces, dots, colons, etc.)
|
||||
# Adding hyphens here because FTS5 can have issues with hyphens followed by wildcards
|
||||
needs_quoting_chars = [" ", ".", ":", ";", ",", "<", ">", "?", "/", "-"]
|
||||
|
||||
# Check if term needs quoting
|
||||
has_problematic = any(c in term for c in problematic_chars)
|
||||
has_spaces_or_special = any(c in term for c in needs_quoting_chars)
|
||||
|
||||
if has_problematic or has_spaces_or_special:
|
||||
# Handle multi-word queries differently from special character queries
|
||||
if " " in term and not any(c in term for c in problematic_chars):
|
||||
# Check if any individual word contains special characters that need quoting
|
||||
words = term.strip().split()
|
||||
has_special_in_words = any(
|
||||
any(c in word for c in needs_quoting_chars if c != " ") for word in words
|
||||
)
|
||||
|
||||
if not has_special_in_words:
|
||||
# For multi-word queries with simple words (like "emoji unicode"),
|
||||
# use boolean AND to handle word order variations
|
||||
if is_prefix:
|
||||
# Add prefix wildcard to each word for better matching
|
||||
prepared_words = [f"{word}*" for word in words if word]
|
||||
else:
|
||||
prepared_words = words
|
||||
term = " AND ".join(prepared_words)
|
||||
else:
|
||||
# If any word has special characters, quote the entire phrase
|
||||
escaped_term = term.replace('"', '""')
|
||||
if is_prefix and not ("/" in term and term.endswith(".md")):
|
||||
term = f'"{escaped_term}"*'
|
||||
else:
|
||||
term = f'"{escaped_term}"'
|
||||
else:
|
||||
# For terms with problematic characters or file paths, use exact phrase matching
|
||||
# Escape any existing quotes by doubling them
|
||||
escaped_term = term.replace('"', '""')
|
||||
# Quote the entire term to handle special characters safely
|
||||
if is_prefix and not ("/" in term and term.endswith(".md")):
|
||||
# For search terms (not file paths), add prefix matching
|
||||
term = f'"{escaped_term}"*'
|
||||
else:
|
||||
# For file paths, use exact matching
|
||||
term = f'"{escaped_term}"'
|
||||
elif is_prefix:
|
||||
# Only add wildcard for simple terms without special characters
|
||||
term = f"{term}*"
|
||||
|
||||
return term
|
||||
|
||||
def _prepare_search_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a search term for FTS5 query.
|
||||
|
||||
Args:
|
||||
term: The search term to prepare
|
||||
is_prefix: Whether to add prefix search capability (* suffix)
|
||||
|
||||
For FTS5:
|
||||
- Boolean operators (AND, OR, NOT) are preserved for complex queries
|
||||
- Terms with FTS5 special characters are quoted to prevent syntax errors
|
||||
- Simple terms get prefix wildcards for better matching
|
||||
"""
|
||||
# Check for explicit boolean operators - if present, process as Boolean query
|
||||
boolean_operators = [" AND ", " OR ", " NOT "]
|
||||
if any(op in f" {term} " for op in boolean_operators):
|
||||
return self._prepare_boolean_query(term)
|
||||
|
||||
# For non-Boolean queries, use the single term preparation logic
|
||||
return self._prepare_single_term(term, is_prefix)
|
||||
async def init_search_index(self) -> None:
|
||||
"""Initialize the search index schema."""
|
||||
...
|
||||
|
||||
async def search(
|
||||
self,
|
||||
@@ -373,267 +43,52 @@ class SearchRepository:
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content with fuzzy matching."""
|
||||
conditions = []
|
||||
params = {}
|
||||
order_by_clause = ""
|
||||
"""Search across indexed content."""
|
||||
...
|
||||
|
||||
# Handle text search for title and content
|
||||
if search_text:
|
||||
# Skip FTS for wildcard-only queries that would cause "unknown special query" errors
|
||||
if search_text.strip() == "*" or search_text.strip() == "":
|
||||
# For wildcard searches, don't add any text conditions - return all results
|
||||
pass
|
||||
else:
|
||||
# Use _prepare_search_term to handle both Boolean and non-Boolean queries
|
||||
processed_text = self._prepare_search_term(search_text.strip())
|
||||
params["text"] = processed_text
|
||||
conditions.append("(title MATCH :text OR content_stems MATCH :text)")
|
||||
async def index_item(self, search_index_row: SearchIndexRow) -> None:
|
||||
"""Index a single item."""
|
||||
...
|
||||
|
||||
# Handle title match search
|
||||
if title:
|
||||
title_text = self._prepare_search_term(title.strip(), is_prefix=False)
|
||||
params["title_text"] = title_text
|
||||
conditions.append("title MATCH :title_text")
|
||||
async def bulk_index_items(self, search_index_rows: List[SearchIndexRow]) -> None:
|
||||
"""Index multiple items in a batch."""
|
||||
...
|
||||
|
||||
# Handle permalink exact search
|
||||
if permalink:
|
||||
params["permalink"] = permalink
|
||||
conditions.append("permalink = :permalink")
|
||||
async def delete_by_permalink(self, permalink: str) -> None:
|
||||
"""Delete item by permalink."""
|
||||
...
|
||||
|
||||
# Handle permalink match search, supports *
|
||||
if permalink_match:
|
||||
# For GLOB patterns, don't use _prepare_search_term as it will quote slashes
|
||||
# GLOB patterns need to preserve their syntax
|
||||
permalink_text = permalink_match.lower().strip()
|
||||
params["permalink"] = permalink_text
|
||||
if "*" in permalink_match:
|
||||
conditions.append("permalink GLOB :permalink")
|
||||
else:
|
||||
# For exact matches without *, we can use FTS5 MATCH
|
||||
# but only prepare the term if it doesn't look like a path
|
||||
if "/" in permalink_text:
|
||||
conditions.append("permalink = :permalink")
|
||||
else:
|
||||
permalink_text = self._prepare_search_term(permalink_text, is_prefix=False)
|
||||
params["permalink"] = permalink_text
|
||||
conditions.append("permalink MATCH :permalink")
|
||||
async def delete_by_entity_id(self, entity_id: int) -> None:
|
||||
"""Delete items by entity ID."""
|
||||
...
|
||||
|
||||
# Handle entity type filter
|
||||
if search_item_types:
|
||||
type_list = ", ".join(f"'{t.value}'" for t in search_item_types)
|
||||
conditions.append(f"type IN ({type_list})")
|
||||
async def execute_query(self, query, params: dict) -> Result:
|
||||
"""Execute a raw SQL query."""
|
||||
...
|
||||
|
||||
# Handle type filter
|
||||
if types:
|
||||
type_list = ", ".join(f"'{t}'" for t in types)
|
||||
conditions.append(f"json_extract(metadata, '$.entity_type') IN ({type_list})")
|
||||
|
||||
# Handle date filter using datetime() for proper comparison
|
||||
if after_date:
|
||||
params["after_date"] = after_date
|
||||
conditions.append("datetime(created_at) > datetime(:after_date)")
|
||||
def create_search_repository(
|
||||
session_maker: async_sessionmaker[AsyncSession], project_id: int
|
||||
) -> SearchRepository:
|
||||
"""Factory function to create the appropriate search repository based on database backend.
|
||||
|
||||
# order by most recent first
|
||||
order_by_clause = ", updated_at DESC"
|
||||
Args:
|
||||
session_maker: SQLAlchemy async session maker
|
||||
project_id: Project ID for the repository
|
||||
|
||||
# Always filter by project_id
|
||||
params["project_id"] = self.project_id
|
||||
conditions.append("project_id = :project_id")
|
||||
Returns:
|
||||
SearchRepository: Backend-appropriate search repository instance
|
||||
"""
|
||||
config = ConfigManager().config
|
||||
|
||||
# set limit on search query
|
||||
params["limit"] = limit
|
||||
params["offset"] = offset
|
||||
if config.database_backend == DatabaseBackend.POSTGRES:
|
||||
return PostgresSearchRepository(session_maker, project_id=project_id)
|
||||
else:
|
||||
return SQLiteSearchRepository(session_maker, project_id=project_id)
|
||||
|
||||
# Build WHERE clause
|
||||
where_clause = " AND ".join(conditions) if conditions else "1=1"
|
||||
|
||||
sql = f"""
|
||||
SELECT
|
||||
project_id,
|
||||
id,
|
||||
title,
|
||||
permalink,
|
||||
file_path,
|
||||
type,
|
||||
metadata,
|
||||
from_id,
|
||||
to_id,
|
||||
relation_type,
|
||||
entity_id,
|
||||
content_snippet,
|
||||
category,
|
||||
created_at,
|
||||
updated_at,
|
||||
bm25(search_index) as score
|
||||
FROM search_index
|
||||
WHERE {where_clause}
|
||||
ORDER BY score ASC {order_by_clause}
|
||||
LIMIT :limit
|
||||
OFFSET :offset
|
||||
"""
|
||||
|
||||
logger.trace(f"Search {sql} params: {params}")
|
||||
try:
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
except Exception as e:
|
||||
# Handle FTS5 syntax errors and provide user-friendly feedback
|
||||
if "fts5: syntax error" in str(e).lower(): # pragma: no cover
|
||||
logger.warning(f"FTS5 syntax error for search term: {search_text}, error: {e}")
|
||||
# Return empty results rather than crashing
|
||||
return []
|
||||
else:
|
||||
# Re-raise other database errors
|
||||
logger.error(f"Database error during search: {e}")
|
||||
raise
|
||||
|
||||
results = [
|
||||
SearchIndexRow(
|
||||
project_id=self.project_id,
|
||||
id=row.id,
|
||||
title=row.title,
|
||||
permalink=row.permalink,
|
||||
file_path=row.file_path,
|
||||
type=row.type,
|
||||
score=row.score,
|
||||
metadata=json.loads(row.metadata),
|
||||
from_id=row.from_id,
|
||||
to_id=row.to_id,
|
||||
relation_type=row.relation_type,
|
||||
entity_id=row.entity_id,
|
||||
content_snippet=row.content_snippet,
|
||||
category=row.category,
|
||||
created_at=row.created_at,
|
||||
updated_at=row.updated_at,
|
||||
)
|
||||
for row in rows
|
||||
]
|
||||
|
||||
logger.trace(f"Found {len(results)} search results")
|
||||
for r in results:
|
||||
logger.trace(
|
||||
f"Search result: project_id: {r.project_id} type:{r.type} title: {r.title} permalink: {r.permalink} score: {r.score}"
|
||||
)
|
||||
|
||||
return results
|
||||
|
||||
async def index_item(
|
||||
self,
|
||||
search_index_row: SearchIndexRow,
|
||||
):
|
||||
"""Index or update a single item."""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# Delete existing record if any
|
||||
await session.execute(
|
||||
text(
|
||||
"DELETE FROM search_index WHERE permalink = :permalink AND project_id = :project_id"
|
||||
),
|
||||
{"permalink": search_index_row.permalink, "project_id": self.project_id},
|
||||
)
|
||||
|
||||
# Prepare data for insert with project_id
|
||||
insert_data = search_index_row.to_insert()
|
||||
insert_data["project_id"] = self.project_id
|
||||
|
||||
# Insert new record
|
||||
await session.execute(
|
||||
text("""
|
||||
INSERT INTO search_index (
|
||||
id, title, content_stems, content_snippet, permalink, file_path, type, metadata,
|
||||
from_id, to_id, relation_type,
|
||||
entity_id, category,
|
||||
created_at, updated_at,
|
||||
project_id
|
||||
) VALUES (
|
||||
:id, :title, :content_stems, :content_snippet, :permalink, :file_path, :type, :metadata,
|
||||
:from_id, :to_id, :relation_type,
|
||||
:entity_id, :category,
|
||||
:created_at, :updated_at,
|
||||
:project_id
|
||||
)
|
||||
"""),
|
||||
insert_data,
|
||||
)
|
||||
logger.debug(f"indexed row {search_index_row}")
|
||||
await session.commit()
|
||||
|
||||
async def bulk_index_items(self, search_index_rows: List[SearchIndexRow]):
|
||||
"""Index multiple items in a single batch operation.
|
||||
|
||||
Note: This method assumes that any existing records for the entity_id
|
||||
have already been deleted (typically via delete_by_entity_id).
|
||||
|
||||
Args:
|
||||
search_index_rows: List of SearchIndexRow objects to index
|
||||
"""
|
||||
if not search_index_rows:
|
||||
return
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# Prepare all insert data with project_id
|
||||
insert_data_list = []
|
||||
for row in search_index_rows:
|
||||
insert_data = row.to_insert()
|
||||
insert_data["project_id"] = self.project_id
|
||||
insert_data_list.append(insert_data)
|
||||
|
||||
# Batch insert all records using executemany
|
||||
await session.execute(
|
||||
text("""
|
||||
INSERT INTO search_index (
|
||||
id, title, content_stems, content_snippet, permalink, file_path, type, metadata,
|
||||
from_id, to_id, relation_type,
|
||||
entity_id, category,
|
||||
created_at, updated_at,
|
||||
project_id
|
||||
) VALUES (
|
||||
:id, :title, :content_stems, :content_snippet, :permalink, :file_path, :type, :metadata,
|
||||
:from_id, :to_id, :relation_type,
|
||||
:entity_id, :category,
|
||||
:created_at, :updated_at,
|
||||
:project_id
|
||||
)
|
||||
"""),
|
||||
insert_data_list,
|
||||
)
|
||||
logger.debug(f"Bulk indexed {len(search_index_rows)} rows")
|
||||
await session.commit()
|
||||
|
||||
async def delete_by_entity_id(self, entity_id: int):
|
||||
"""Delete an item from the search index by entity_id."""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
await session.execute(
|
||||
text(
|
||||
"DELETE FROM search_index WHERE entity_id = :entity_id AND project_id = :project_id"
|
||||
),
|
||||
{"entity_id": entity_id, "project_id": self.project_id},
|
||||
)
|
||||
await session.commit()
|
||||
|
||||
async def delete_by_permalink(self, permalink: str):
|
||||
"""Delete an item from the search index."""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
await session.execute(
|
||||
text(
|
||||
"DELETE FROM search_index WHERE permalink = :permalink AND project_id = :project_id"
|
||||
),
|
||||
{"permalink": permalink, "project_id": self.project_id},
|
||||
)
|
||||
await session.commit()
|
||||
|
||||
async def execute_query(
|
||||
self,
|
||||
query: Executable,
|
||||
params: Dict[str, Any],
|
||||
) -> Result[Any]:
|
||||
"""Execute a query asynchronously."""
|
||||
# logger.debug(f"Executing query: {query}, params: {params}")
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
start_time = time.perf_counter()
|
||||
result = await session.execute(query, params)
|
||||
end_time = time.perf_counter()
|
||||
elapsed_time = end_time - start_time
|
||||
logger.debug(f"Query executed successfully in {elapsed_time:.2f}s.")
|
||||
return result
|
||||
__all__ = [
|
||||
"SearchRepository",
|
||||
"SearchIndexRow",
|
||||
"create_search_repository",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,241 @@
|
||||
"""Abstract base class for search repository implementations."""
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import Executable, Result, text
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
|
||||
|
||||
class SearchRepositoryBase(ABC):
|
||||
"""Abstract base class for backend-specific search repository implementations.
|
||||
|
||||
This class defines the common interface that all search repositories must implement,
|
||||
regardless of whether they use SQLite FTS5 or Postgres tsvector for full-text search.
|
||||
|
||||
Concrete implementations:
|
||||
- SQLiteSearchRepository: Uses FTS5 virtual tables with MATCH queries
|
||||
- PostgresSearchRepository: Uses tsvector/tsquery with GIN indexes
|
||||
"""
|
||||
|
||||
def __init__(self, session_maker: async_sessionmaker[AsyncSession], project_id: int):
|
||||
"""Initialize with session maker and project_id filter.
|
||||
|
||||
Args:
|
||||
session_maker: SQLAlchemy session maker
|
||||
project_id: Project ID to filter all operations by
|
||||
|
||||
Raises:
|
||||
ValueError: If project_id is None or invalid
|
||||
"""
|
||||
if project_id is None or project_id <= 0: # pragma: no cover
|
||||
raise ValueError("A valid project_id is required for SearchRepository")
|
||||
|
||||
self.session_maker = session_maker
|
||||
self.project_id = project_id
|
||||
|
||||
@abstractmethod
|
||||
async def init_search_index(self) -> None:
|
||||
"""Create or recreate the search index.
|
||||
|
||||
Backend-specific implementations:
|
||||
- SQLite: CREATE VIRTUAL TABLE using FTS5
|
||||
- Postgres: CREATE TABLE with tsvector column and GIN indexes
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def _prepare_search_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a search term for backend-specific query syntax.
|
||||
|
||||
Args:
|
||||
term: The search term to prepare
|
||||
is_prefix: Whether to add prefix search capability
|
||||
|
||||
Returns:
|
||||
Formatted search term for the backend
|
||||
|
||||
Backend-specific implementations:
|
||||
- SQLite: Quotes FTS5 special characters, adds * wildcards
|
||||
- Postgres: Converts to tsquery syntax with :* prefix operator
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
async def search(
|
||||
self,
|
||||
search_text: Optional[str] = None,
|
||||
permalink: Optional[str] = None,
|
||||
permalink_match: Optional[str] = None,
|
||||
title: Optional[str] = None,
|
||||
types: Optional[List[str]] = None,
|
||||
after_date: Optional[datetime] = None,
|
||||
search_item_types: Optional[List[SearchItemType]] = None,
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content.
|
||||
|
||||
Args:
|
||||
search_text: Full-text search across title and content
|
||||
permalink: Exact permalink match
|
||||
permalink_match: Permalink pattern match (supports *)
|
||||
title: Title search
|
||||
types: Filter by entity types (from metadata.entity_type)
|
||||
after_date: Filter by created_at > after_date
|
||||
search_item_types: Filter by SearchItemType (ENTITY, OBSERVATION, RELATION)
|
||||
limit: Maximum results to return
|
||||
offset: Number of results to skip
|
||||
|
||||
Returns:
|
||||
List of SearchIndexRow results with relevance scores
|
||||
|
||||
Backend-specific implementations:
|
||||
- SQLite: Uses MATCH operator and bm25() for scoring
|
||||
- Postgres: Uses @@ operator and ts_rank() for scoring
|
||||
"""
|
||||
pass
|
||||
|
||||
async def index_item(self, search_index_row: SearchIndexRow) -> None:
|
||||
"""Index or update a single item.
|
||||
|
||||
This implementation is shared across backends as it uses standard SQL INSERT.
|
||||
"""
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# Delete existing record if any
|
||||
await session.execute(
|
||||
text(
|
||||
"DELETE FROM search_index WHERE permalink = :permalink AND project_id = :project_id"
|
||||
),
|
||||
{"permalink": search_index_row.permalink, "project_id": self.project_id},
|
||||
)
|
||||
|
||||
# When using text() raw SQL, always serialize JSON to string
|
||||
# Both SQLite (TEXT) and Postgres (JSONB) accept JSON strings in raw SQL
|
||||
# The database driver/column type will handle conversion
|
||||
insert_data = search_index_row.to_insert(serialize_json=True)
|
||||
insert_data["project_id"] = self.project_id
|
||||
|
||||
# Insert new record
|
||||
await session.execute(
|
||||
text("""
|
||||
INSERT INTO search_index (
|
||||
id, title, content_stems, content_snippet, permalink, file_path, type, metadata,
|
||||
from_id, to_id, relation_type,
|
||||
entity_id, category,
|
||||
created_at, updated_at,
|
||||
project_id
|
||||
) VALUES (
|
||||
:id, :title, :content_stems, :content_snippet, :permalink, :file_path, :type, :metadata,
|
||||
:from_id, :to_id, :relation_type,
|
||||
:entity_id, :category,
|
||||
:created_at, :updated_at,
|
||||
:project_id
|
||||
)
|
||||
"""),
|
||||
insert_data,
|
||||
)
|
||||
logger.debug(f"indexed row {search_index_row}")
|
||||
await session.commit()
|
||||
|
||||
async def bulk_index_items(self, search_index_rows: List[SearchIndexRow]) -> None:
|
||||
"""Index multiple items in a single batch operation.
|
||||
|
||||
This implementation is shared across backends as it uses standard SQL INSERT.
|
||||
|
||||
Note: This method assumes that any existing records for the entity_id
|
||||
have already been deleted (typically via delete_by_entity_id).
|
||||
|
||||
Args:
|
||||
search_index_rows: List of SearchIndexRow objects to index
|
||||
"""
|
||||
|
||||
if not search_index_rows:
|
||||
return
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# When using text() raw SQL, always serialize JSON to string
|
||||
# Both SQLite (TEXT) and Postgres (JSONB) accept JSON strings in raw SQL
|
||||
# The database driver/column type will handle conversion
|
||||
insert_data_list = []
|
||||
for row in search_index_rows:
|
||||
insert_data = row.to_insert(serialize_json=True)
|
||||
insert_data["project_id"] = self.project_id
|
||||
insert_data_list.append(insert_data)
|
||||
|
||||
# Batch insert all records using executemany
|
||||
await session.execute(
|
||||
text("""
|
||||
INSERT INTO search_index (
|
||||
id, title, content_stems, content_snippet, permalink, file_path, type, metadata,
|
||||
from_id, to_id, relation_type,
|
||||
entity_id, category,
|
||||
created_at, updated_at,
|
||||
project_id
|
||||
) VALUES (
|
||||
:id, :title, :content_stems, :content_snippet, :permalink, :file_path, :type, :metadata,
|
||||
:from_id, :to_id, :relation_type,
|
||||
:entity_id, :category,
|
||||
:created_at, :updated_at,
|
||||
:project_id
|
||||
)
|
||||
"""),
|
||||
insert_data_list,
|
||||
)
|
||||
logger.debug(f"Bulk indexed {len(search_index_rows)} rows")
|
||||
await session.commit()
|
||||
|
||||
async def delete_by_entity_id(self, entity_id: int) -> None:
|
||||
"""Delete all search index entries for an entity.
|
||||
|
||||
This implementation is shared across backends as it uses standard SQL DELETE.
|
||||
"""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
await session.execute(
|
||||
text(
|
||||
"DELETE FROM search_index WHERE entity_id = :entity_id AND project_id = :project_id"
|
||||
),
|
||||
{"entity_id": entity_id, "project_id": self.project_id},
|
||||
)
|
||||
await session.commit()
|
||||
|
||||
async def delete_by_permalink(self, permalink: str) -> None:
|
||||
"""Delete a search index entry by permalink.
|
||||
|
||||
This implementation is shared across backends as it uses standard SQL DELETE.
|
||||
"""
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
await session.execute(
|
||||
text(
|
||||
"DELETE FROM search_index WHERE permalink = :permalink AND project_id = :project_id"
|
||||
),
|
||||
{"permalink": permalink, "project_id": self.project_id},
|
||||
)
|
||||
await session.commit()
|
||||
|
||||
async def execute_query(
|
||||
self,
|
||||
query: Executable,
|
||||
params: Dict[str, Any],
|
||||
) -> Result[Any]:
|
||||
"""Execute a query asynchronously.
|
||||
|
||||
This implementation is shared across backends for utility query execution.
|
||||
"""
|
||||
import time
|
||||
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
start_time = time.perf_counter()
|
||||
result = await session.execute(query, params)
|
||||
end_time = time.perf_counter()
|
||||
elapsed_time = end_time - start_time
|
||||
logger.debug(f"Query executed successfully in {elapsed_time:.2f}s.")
|
||||
return result
|
||||
@@ -0,0 +1,439 @@
|
||||
"""SQLite FTS5-based search repository implementation."""
|
||||
|
||||
import json
|
||||
import re
|
||||
from datetime import datetime
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
from basic_memory import db
|
||||
from basic_memory.models.search import CREATE_SEARCH_INDEX
|
||||
from basic_memory.repository.search_index_row import SearchIndexRow
|
||||
from basic_memory.repository.search_repository_base import SearchRepositoryBase
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
|
||||
|
||||
class SQLiteSearchRepository(SearchRepositoryBase):
|
||||
"""SQLite FTS5 implementation of search repository.
|
||||
|
||||
Uses SQLite's FTS5 virtual tables for full-text search with:
|
||||
- MATCH operator for queries
|
||||
- bm25() function for relevance scoring
|
||||
- Special character quoting for syntax safety
|
||||
- Prefix wildcard matching with *
|
||||
"""
|
||||
|
||||
async def init_search_index(self):
|
||||
"""Create FTS5 virtual table for search.
|
||||
|
||||
Note: Drops any existing search_index table first to ensure FTS5 virtual table creation.
|
||||
This is necessary because Base.metadata.create_all() might create a regular table.
|
||||
"""
|
||||
logger.info("Initializing SQLite FTS5 search index")
|
||||
try:
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
# Drop any existing regular or virtual table first
|
||||
await session.execute(text("DROP TABLE IF EXISTS search_index"))
|
||||
# Create FTS5 virtual table
|
||||
await session.execute(CREATE_SEARCH_INDEX)
|
||||
await session.commit()
|
||||
except Exception as e: # pragma: no cover
|
||||
logger.error(f"Error initializing search index: {e}")
|
||||
raise e
|
||||
|
||||
def _prepare_boolean_query(self, query: str) -> str:
|
||||
"""Prepare a Boolean query by quoting individual terms while preserving operators.
|
||||
|
||||
Args:
|
||||
query: A Boolean query like "tier1-test AND unicode" or "(hello OR world) NOT test"
|
||||
|
||||
Returns:
|
||||
A properly formatted Boolean query with quoted terms that need quoting
|
||||
"""
|
||||
# Define Boolean operators and their boundaries
|
||||
boolean_pattern = r"(\bAND\b|\bOR\b|\bNOT\b)"
|
||||
|
||||
# Split the query by Boolean operators, keeping the operators
|
||||
parts = re.split(boolean_pattern, query)
|
||||
|
||||
processed_parts = []
|
||||
for part in parts:
|
||||
part = part.strip()
|
||||
if not part:
|
||||
continue
|
||||
|
||||
# If it's a Boolean operator, keep it as is
|
||||
if part in ["AND", "OR", "NOT"]:
|
||||
processed_parts.append(part)
|
||||
else:
|
||||
# Handle parentheses specially - they should be preserved for grouping
|
||||
if "(" in part or ")" in part:
|
||||
# Parse parenthetical expressions carefully
|
||||
processed_part = self._prepare_parenthetical_term(part)
|
||||
processed_parts.append(processed_part)
|
||||
else:
|
||||
# This is a search term - for Boolean queries, don't add prefix wildcards
|
||||
prepared_term = self._prepare_single_term(part, is_prefix=False)
|
||||
processed_parts.append(prepared_term)
|
||||
|
||||
return " ".join(processed_parts)
|
||||
|
||||
def _prepare_parenthetical_term(self, term: str) -> str:
|
||||
"""Prepare a term that contains parentheses, preserving the parentheses for grouping.
|
||||
|
||||
Args:
|
||||
term: A term that may contain parentheses like "(hello" or "world)" or "(hello OR world)"
|
||||
|
||||
Returns:
|
||||
A properly formatted term with parentheses preserved
|
||||
"""
|
||||
# Handle terms that start/end with parentheses but may contain quotable content
|
||||
result = ""
|
||||
i = 0
|
||||
while i < len(term):
|
||||
if term[i] in "()":
|
||||
# Preserve parentheses as-is
|
||||
result += term[i]
|
||||
i += 1
|
||||
else:
|
||||
# Find the next parenthesis or end of string
|
||||
start = i
|
||||
while i < len(term) and term[i] not in "()":
|
||||
i += 1
|
||||
|
||||
# Extract the content between parentheses
|
||||
content = term[start:i].strip()
|
||||
if content:
|
||||
# Only quote if it actually needs quoting (has hyphens, special chars, etc)
|
||||
# but don't quote if it's just simple words
|
||||
if self._needs_quoting(content):
|
||||
escaped_content = content.replace('"', '""')
|
||||
result += f'"{escaped_content}"'
|
||||
else:
|
||||
result += content
|
||||
|
||||
return result
|
||||
|
||||
def _needs_quoting(self, term: str) -> bool:
|
||||
"""Check if a term needs to be quoted for FTS5 safety.
|
||||
|
||||
Args:
|
||||
term: The term to check
|
||||
|
||||
Returns:
|
||||
True if the term should be quoted
|
||||
"""
|
||||
if not term or not term.strip():
|
||||
return False
|
||||
|
||||
# Characters that indicate we should quote (excluding parentheses which are valid syntax)
|
||||
needs_quoting_chars = [
|
||||
" ",
|
||||
".",
|
||||
":",
|
||||
";",
|
||||
",",
|
||||
"<",
|
||||
">",
|
||||
"?",
|
||||
"/",
|
||||
"-",
|
||||
"'",
|
||||
'"',
|
||||
"[",
|
||||
"]",
|
||||
"{",
|
||||
"}",
|
||||
"+",
|
||||
"!",
|
||||
"@",
|
||||
"#",
|
||||
"$",
|
||||
"%",
|
||||
"^",
|
||||
"&",
|
||||
"=",
|
||||
"|",
|
||||
"\\",
|
||||
"~",
|
||||
"`",
|
||||
]
|
||||
|
||||
return any(c in term for c in needs_quoting_chars)
|
||||
|
||||
def _prepare_single_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a single search term (no Boolean operators).
|
||||
|
||||
Args:
|
||||
term: A single search term
|
||||
is_prefix: Whether to add prefix search capability (* suffix)
|
||||
|
||||
Returns:
|
||||
A properly formatted single term
|
||||
"""
|
||||
if not term or not term.strip():
|
||||
return term
|
||||
|
||||
term = term.strip()
|
||||
|
||||
# Check if term is already a proper wildcard pattern (alphanumeric + *)
|
||||
# e.g., "hello*", "test*world" - these should be left alone
|
||||
if "*" in term and all(c.isalnum() or c in "*_-" for c in term):
|
||||
return term
|
||||
|
||||
# Characters that can cause FTS5 syntax errors when used as operators
|
||||
# We're more conservative here - only quote when we detect problematic patterns
|
||||
problematic_chars = [
|
||||
'"',
|
||||
"'",
|
||||
"(",
|
||||
")",
|
||||
"[",
|
||||
"]",
|
||||
"{",
|
||||
"}",
|
||||
"+",
|
||||
"!",
|
||||
"@",
|
||||
"#",
|
||||
"$",
|
||||
"%",
|
||||
"^",
|
||||
"&",
|
||||
"=",
|
||||
"|",
|
||||
"\\",
|
||||
"~",
|
||||
"`",
|
||||
]
|
||||
|
||||
# Characters that indicate we should quote (spaces, dots, colons, etc.)
|
||||
# Adding hyphens here because FTS5 can have issues with hyphens followed by wildcards
|
||||
needs_quoting_chars = [" ", ".", ":", ";", ",", "<", ">", "?", "/", "-"]
|
||||
|
||||
# Check if term needs quoting
|
||||
has_problematic = any(c in term for c in problematic_chars)
|
||||
has_spaces_or_special = any(c in term for c in needs_quoting_chars)
|
||||
|
||||
if has_problematic or has_spaces_or_special:
|
||||
# Handle multi-word queries differently from special character queries
|
||||
if " " in term and not any(c in term for c in problematic_chars):
|
||||
# Check if any individual word contains special characters that need quoting
|
||||
words = term.strip().split()
|
||||
has_special_in_words = any(
|
||||
any(c in word for c in needs_quoting_chars if c != " ") for word in words
|
||||
)
|
||||
|
||||
if not has_special_in_words:
|
||||
# For multi-word queries with simple words (like "emoji unicode"),
|
||||
# use boolean AND to handle word order variations
|
||||
if is_prefix:
|
||||
# Add prefix wildcard to each word for better matching
|
||||
prepared_words = [f"{word}*" for word in words if word]
|
||||
else:
|
||||
prepared_words = words
|
||||
term = " AND ".join(prepared_words)
|
||||
else:
|
||||
# If any word has special characters, quote the entire phrase
|
||||
escaped_term = term.replace('"', '""')
|
||||
if is_prefix and not ("/" in term and term.endswith(".md")):
|
||||
term = f'"{escaped_term}"*'
|
||||
else:
|
||||
term = f'"{escaped_term}"'
|
||||
else:
|
||||
# For terms with problematic characters or file paths, use exact phrase matching
|
||||
# Escape any existing quotes by doubling them
|
||||
escaped_term = term.replace('"', '""')
|
||||
# Quote the entire term to handle special characters safely
|
||||
if is_prefix and not ("/" in term and term.endswith(".md")):
|
||||
# For search terms (not file paths), add prefix matching
|
||||
term = f'"{escaped_term}"*'
|
||||
else:
|
||||
# For file paths, use exact matching
|
||||
term = f'"{escaped_term}"'
|
||||
elif is_prefix:
|
||||
# Only add wildcard for simple terms without special characters
|
||||
term = f"{term}*"
|
||||
|
||||
return term
|
||||
|
||||
def _prepare_search_term(self, term: str, is_prefix: bool = True) -> str:
|
||||
"""Prepare a search term for FTS5 query.
|
||||
|
||||
Args:
|
||||
term: The search term to prepare
|
||||
is_prefix: Whether to add prefix search capability (* suffix)
|
||||
|
||||
For FTS5:
|
||||
- Boolean operators (AND, OR, NOT) are preserved for complex queries
|
||||
- Terms with FTS5 special characters are quoted to prevent syntax errors
|
||||
- Simple terms get prefix wildcards for better matching
|
||||
"""
|
||||
# Check for explicit boolean operators - if present, process as Boolean query
|
||||
boolean_operators = [" AND ", " OR ", " NOT "]
|
||||
if any(op in f" {term} " for op in boolean_operators):
|
||||
return self._prepare_boolean_query(term)
|
||||
|
||||
# For non-Boolean queries, use the single term preparation logic
|
||||
return self._prepare_single_term(term, is_prefix)
|
||||
|
||||
async def search(
|
||||
self,
|
||||
search_text: Optional[str] = None,
|
||||
permalink: Optional[str] = None,
|
||||
permalink_match: Optional[str] = None,
|
||||
title: Optional[str] = None,
|
||||
types: Optional[List[str]] = None,
|
||||
after_date: Optional[datetime] = None,
|
||||
search_item_types: Optional[List[SearchItemType]] = None,
|
||||
limit: int = 10,
|
||||
offset: int = 0,
|
||||
) -> List[SearchIndexRow]:
|
||||
"""Search across all indexed content using SQLite FTS5."""
|
||||
conditions = []
|
||||
params = {}
|
||||
order_by_clause = ""
|
||||
|
||||
# Handle text search for title and content
|
||||
if search_text:
|
||||
# Skip FTS for wildcard-only queries that would cause "unknown special query" errors
|
||||
if search_text.strip() == "*" or search_text.strip() == "":
|
||||
# For wildcard searches, don't add any text conditions - return all results
|
||||
pass
|
||||
else:
|
||||
# Use _prepare_search_term to handle both Boolean and non-Boolean queries
|
||||
processed_text = self._prepare_search_term(search_text.strip())
|
||||
params["text"] = processed_text
|
||||
conditions.append("(title MATCH :text OR content_stems MATCH :text)")
|
||||
|
||||
# Handle title match search
|
||||
if title:
|
||||
title_text = self._prepare_search_term(title.strip(), is_prefix=False)
|
||||
params["title_text"] = title_text
|
||||
conditions.append("title MATCH :title_text")
|
||||
|
||||
# Handle permalink exact search
|
||||
if permalink:
|
||||
params["permalink"] = permalink
|
||||
conditions.append("permalink = :permalink")
|
||||
|
||||
# Handle permalink match search, supports *
|
||||
if permalink_match:
|
||||
# For GLOB patterns, don't use _prepare_search_term as it will quote slashes
|
||||
# GLOB patterns need to preserve their syntax
|
||||
permalink_text = permalink_match.lower().strip()
|
||||
params["permalink"] = permalink_text
|
||||
if "*" in permalink_match:
|
||||
conditions.append("permalink GLOB :permalink")
|
||||
else:
|
||||
# For exact matches without *, we can use FTS5 MATCH
|
||||
# but only prepare the term if it doesn't look like a path
|
||||
if "/" in permalink_text:
|
||||
conditions.append("permalink = :permalink")
|
||||
else:
|
||||
permalink_text = self._prepare_search_term(permalink_text, is_prefix=False)
|
||||
params["permalink"] = permalink_text
|
||||
conditions.append("permalink MATCH :permalink")
|
||||
|
||||
# Handle entity type filter
|
||||
if search_item_types:
|
||||
type_list = ", ".join(f"'{t.value}'" for t in search_item_types)
|
||||
conditions.append(f"type IN ({type_list})")
|
||||
|
||||
# Handle type filter
|
||||
if types:
|
||||
type_list = ", ".join(f"'{t}'" for t in types)
|
||||
conditions.append(f"json_extract(metadata, '$.entity_type') IN ({type_list})")
|
||||
|
||||
# Handle date filter using datetime() for proper comparison
|
||||
if after_date:
|
||||
params["after_date"] = after_date
|
||||
conditions.append("datetime(created_at) > datetime(:after_date)")
|
||||
|
||||
# order by most recent first
|
||||
order_by_clause = ", updated_at DESC"
|
||||
|
||||
# Always filter by project_id
|
||||
params["project_id"] = self.project_id
|
||||
conditions.append("project_id = :project_id")
|
||||
|
||||
# set limit on search query
|
||||
params["limit"] = limit
|
||||
params["offset"] = offset
|
||||
|
||||
# Build WHERE clause
|
||||
where_clause = " AND ".join(conditions) if conditions else "1=1"
|
||||
|
||||
sql = f"""
|
||||
SELECT
|
||||
project_id,
|
||||
id,
|
||||
title,
|
||||
permalink,
|
||||
file_path,
|
||||
type,
|
||||
metadata,
|
||||
from_id,
|
||||
to_id,
|
||||
relation_type,
|
||||
entity_id,
|
||||
content_snippet,
|
||||
category,
|
||||
created_at,
|
||||
updated_at,
|
||||
bm25(search_index) as score
|
||||
FROM search_index
|
||||
WHERE {where_clause}
|
||||
ORDER BY score ASC {order_by_clause}
|
||||
LIMIT :limit
|
||||
OFFSET :offset
|
||||
"""
|
||||
|
||||
logger.trace(f"Search {sql} params: {params}")
|
||||
try:
|
||||
async with db.scoped_session(self.session_maker) as session:
|
||||
result = await session.execute(text(sql), params)
|
||||
rows = result.fetchall()
|
||||
except Exception as e:
|
||||
# Handle FTS5 syntax errors and provide user-friendly feedback
|
||||
if "fts5: syntax error" in str(e).lower(): # pragma: no cover
|
||||
logger.warning(f"FTS5 syntax error for search term: {search_text}, error: {e}")
|
||||
# Return empty results rather than crashing
|
||||
return []
|
||||
else:
|
||||
# Re-raise other database errors
|
||||
logger.error(f"Database error during search: {e}")
|
||||
raise
|
||||
|
||||
results = [
|
||||
SearchIndexRow(
|
||||
project_id=self.project_id,
|
||||
id=row.id,
|
||||
title=row.title,
|
||||
permalink=row.permalink,
|
||||
file_path=row.file_path,
|
||||
type=row.type,
|
||||
score=row.score,
|
||||
metadata=json.loads(row.metadata) if row.metadata else {},
|
||||
from_id=row.from_id,
|
||||
to_id=row.to_id,
|
||||
relation_type=row.relation_type,
|
||||
entity_id=row.entity_id,
|
||||
content_snippet=row.content_snippet,
|
||||
category=row.category,
|
||||
created_at=row.created_at,
|
||||
updated_at=row.updated_at,
|
||||
)
|
||||
for row in rows
|
||||
]
|
||||
|
||||
logger.trace(f"Found {len(results)} search results")
|
||||
for r in results:
|
||||
logger.trace(
|
||||
f"Search result: project_id: {r.project_id} type:{r.type} title: {r.title} permalink: {r.permalink} score: {r.score}"
|
||||
)
|
||||
|
||||
return results
|
||||
@@ -21,13 +21,38 @@ from typing import List, Optional, Annotated, Dict
|
||||
from annotated_types import MinLen, MaxLen
|
||||
from dateparser import parse
|
||||
|
||||
from pydantic import BaseModel, BeforeValidator, Field, model_validator
|
||||
from pydantic import BaseModel, BeforeValidator, Field, model_validator, computed_field
|
||||
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.file_utils import sanitize_for_filename, sanitize_for_folder
|
||||
from basic_memory.utils import generate_permalink
|
||||
|
||||
|
||||
def has_valid_file_extension(filename: str) -> bool:
|
||||
"""Check if a filename has a valid file extension recognized by mimetypes.
|
||||
|
||||
This is used to determine whether to split the extension when processing
|
||||
titles in kebab_filenames mode. Prevents treating periods in version numbers
|
||||
or decimals as file extensions.
|
||||
|
||||
Args:
|
||||
filename: The filename to check
|
||||
|
||||
Returns:
|
||||
True if the filename has a recognized file extension, False otherwise
|
||||
|
||||
Examples:
|
||||
>>> has_valid_file_extension("document.md")
|
||||
True
|
||||
>>> has_valid_file_extension("Version 2.0.0")
|
||||
False
|
||||
>>> has_valid_file_extension("image.png")
|
||||
True
|
||||
"""
|
||||
mime_type, _ = mimetypes.guess_type(filename)
|
||||
return mime_type is not None
|
||||
|
||||
|
||||
def to_snake_case(name: str) -> str:
|
||||
"""Convert a string to snake_case.
|
||||
|
||||
@@ -232,12 +257,17 @@ class Entity(BaseModel):
|
||||
use_kebab_case = app_config.kebab_filenames
|
||||
|
||||
if use_kebab_case:
|
||||
fixed_title = generate_permalink(file_path=fixed_title, split_extension=False)
|
||||
# Convert to kebab-case: lowercase with hyphens, preserving periods in version numbers
|
||||
# generate_permalink() uses mimetypes to detect real file extensions and only splits
|
||||
# them off, avoiding misinterpreting periods in version numbers as extensions
|
||||
has_extension = has_valid_file_extension(fixed_title)
|
||||
fixed_title = generate_permalink(file_path=fixed_title, split_extension=has_extension)
|
||||
|
||||
return fixed_title
|
||||
|
||||
@computed_field
|
||||
@property
|
||||
def file_path(self):
|
||||
def file_path(self) -> str:
|
||||
"""Get the file path for this entity based on its permalink."""
|
||||
safe_title = self.safe_title
|
||||
if self.content_type == "text/markdown":
|
||||
|
||||
@@ -124,6 +124,7 @@ class EntitySummary(BaseModel):
|
||||
"""Simplified entity representation."""
|
||||
|
||||
type: Literal["entity"] = "entity"
|
||||
entity_id: int # Database ID for v2 API consistency
|
||||
permalink: Optional[str]
|
||||
title: str
|
||||
content: Optional[str] = None
|
||||
@@ -141,12 +142,16 @@ class RelationSummary(BaseModel):
|
||||
"""Simplified relation representation."""
|
||||
|
||||
type: Literal["relation"] = "relation"
|
||||
relation_id: int # Database ID for v2 API consistency
|
||||
entity_id: Optional[int] = None # ID of the entity this relation belongs to
|
||||
title: str
|
||||
file_path: str
|
||||
permalink: str
|
||||
relation_type: str
|
||||
from_entity: Optional[str] = None
|
||||
from_entity_id: Optional[int] = None # ID of source entity
|
||||
to_entity: Optional[str] = None
|
||||
to_entity_id: Optional[int] = None # ID of target entity
|
||||
created_at: Annotated[
|
||||
datetime, Field(json_schema_extra={"type": "string", "format": "date-time"})
|
||||
]
|
||||
@@ -160,6 +165,8 @@ class ObservationSummary(BaseModel):
|
||||
"""Simplified observation representation."""
|
||||
|
||||
type: Literal["observation"] = "observation"
|
||||
observation_id: int # Database ID for v2 API consistency
|
||||
entity_id: Optional[int] = None # ID of the entity this observation belongs to
|
||||
title: str
|
||||
file_path: str
|
||||
permalink: str
|
||||
|
||||
@@ -173,6 +173,7 @@ class ProjectWatchStatus(BaseModel):
|
||||
class ProjectItem(BaseModel):
|
||||
"""Simple representation of a project."""
|
||||
|
||||
id: int
|
||||
name: str
|
||||
path: str
|
||||
is_default: bool = False
|
||||
|
||||
@@ -97,6 +97,11 @@ class SearchResult(BaseModel):
|
||||
|
||||
metadata: Optional[dict] = None
|
||||
|
||||
# IDs for v2 API consistency
|
||||
entity_id: Optional[int] = None # Entity ID (always present for entities)
|
||||
observation_id: Optional[int] = None # Observation ID (for observation results)
|
||||
relation_id: Optional[int] = None # Relation ID (for relation results)
|
||||
|
||||
# Type-specific fields
|
||||
category: Optional[str] = None # For observations
|
||||
from_entity: Optional[Permalink] = None # For relations
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
"""V2 API schemas - ID-based entity references."""
|
||||
|
||||
from basic_memory.schemas.v2.entity import (
|
||||
EntityResolveRequest,
|
||||
EntityResolveResponse,
|
||||
EntityResponseV2,
|
||||
MoveEntityRequestV2,
|
||||
)
|
||||
from basic_memory.schemas.v2.resource import (
|
||||
CreateResourceRequest,
|
||||
UpdateResourceRequest,
|
||||
ResourceResponse,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"EntityResolveRequest",
|
||||
"EntityResolveResponse",
|
||||
"EntityResponseV2",
|
||||
"MoveEntityRequestV2",
|
||||
"CreateResourceRequest",
|
||||
"UpdateResourceRequest",
|
||||
"ResourceResponse",
|
||||
]
|
||||
@@ -0,0 +1,96 @@
|
||||
"""V2 entity schemas with ID-first design."""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Literal, Optional
|
||||
|
||||
from pydantic import BaseModel, Field, ConfigDict
|
||||
|
||||
from basic_memory.schemas.response import ObservationResponse, RelationResponse
|
||||
|
||||
|
||||
class EntityResolveRequest(BaseModel):
|
||||
"""Request to resolve a string identifier to an entity ID.
|
||||
|
||||
Supports resolution of:
|
||||
- Permalinks (e.g., "specs/search")
|
||||
- Titles (e.g., "Search Specification")
|
||||
- File paths (e.g., "specs/search.md")
|
||||
"""
|
||||
|
||||
identifier: str = Field(
|
||||
...,
|
||||
description="Entity identifier to resolve (permalink, title, or file path)",
|
||||
min_length=1,
|
||||
max_length=500,
|
||||
)
|
||||
|
||||
|
||||
class EntityResolveResponse(BaseModel):
|
||||
"""Response from identifier resolution.
|
||||
|
||||
Returns the entity ID and associated metadata for the resolved entity.
|
||||
"""
|
||||
|
||||
entity_id: int = Field(..., description="Numeric entity ID (primary identifier)")
|
||||
permalink: Optional[str] = Field(None, description="Entity permalink")
|
||||
file_path: str = Field(..., description="Relative file path")
|
||||
title: str = Field(..., description="Entity title")
|
||||
resolution_method: Literal["id", "permalink", "title", "path", "search"] = Field(
|
||||
..., description="How the identifier was resolved"
|
||||
)
|
||||
|
||||
|
||||
class MoveEntityRequestV2(BaseModel):
|
||||
"""V2 request schema for moving an entity to a new file location.
|
||||
|
||||
In V2 API, the entity ID is provided in the URL path, so this request
|
||||
only needs the destination path.
|
||||
"""
|
||||
|
||||
destination_path: str = Field(
|
||||
...,
|
||||
description="New file path for the entity (relative to project root)",
|
||||
min_length=1,
|
||||
max_length=500,
|
||||
)
|
||||
|
||||
|
||||
class EntityResponseV2(BaseModel):
|
||||
"""V2 entity response with ID as the primary field.
|
||||
|
||||
This response format emphasizes the entity ID as the primary identifier,
|
||||
with all other fields (permalink, file_path) as secondary metadata.
|
||||
"""
|
||||
|
||||
# ID first - this is the primary identifier in v2
|
||||
id: int = Field(..., description="Numeric entity ID (primary identifier)")
|
||||
|
||||
# Core entity fields
|
||||
title: str = Field(..., description="Entity title")
|
||||
entity_type: str = Field(..., description="Entity type")
|
||||
content_type: str = Field(default="text/markdown", description="Content MIME type")
|
||||
|
||||
# Secondary identifiers (for compatibility and convenience)
|
||||
permalink: Optional[str] = Field(None, description="Entity permalink (may change)")
|
||||
file_path: str = Field(..., description="Relative file path (may change)")
|
||||
|
||||
# Content and metadata
|
||||
content: Optional[str] = Field(None, description="Entity content")
|
||||
entity_metadata: Optional[Dict] = Field(None, description="Entity metadata")
|
||||
|
||||
# Relationships
|
||||
observations: List[ObservationResponse] = Field(
|
||||
default_factory=list, description="Entity observations"
|
||||
)
|
||||
relations: List[RelationResponse] = Field(default_factory=list, description="Entity relations")
|
||||
|
||||
# Timestamps
|
||||
created_at: datetime = Field(..., description="Creation timestamp")
|
||||
updated_at: datetime = Field(..., description="Last update timestamp")
|
||||
|
||||
# V2-specific metadata
|
||||
api_version: Literal["v2"] = Field(
|
||||
default="v2", description="API version (always 'v2' for this response)"
|
||||
)
|
||||
|
||||
model_config = ConfigDict(from_attributes=True)
|
||||
@@ -0,0 +1,46 @@
|
||||
"""V2 resource schemas for file content operations."""
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class CreateResourceRequest(BaseModel):
|
||||
"""Request to create a new resource file.
|
||||
|
||||
File path is required for new resources since we need to know where
|
||||
to create the file.
|
||||
"""
|
||||
|
||||
file_path: str = Field(
|
||||
...,
|
||||
description="Path to create the file, relative to project root",
|
||||
min_length=1,
|
||||
max_length=500,
|
||||
)
|
||||
content: str = Field(..., description="File content to write")
|
||||
|
||||
|
||||
class UpdateResourceRequest(BaseModel):
|
||||
"""Request to update an existing resource by entity ID.
|
||||
|
||||
Only content is required - the file path is already known from the entity.
|
||||
Optionally can update the file_path to move the file.
|
||||
"""
|
||||
|
||||
content: str = Field(..., description="File content to write")
|
||||
file_path: str | None = Field(
|
||||
None,
|
||||
description="Optional new file path to move the resource",
|
||||
min_length=1,
|
||||
max_length=500,
|
||||
)
|
||||
|
||||
|
||||
class ResourceResponse(BaseModel):
|
||||
"""Response from resource operations."""
|
||||
|
||||
entity_id: int = Field(..., description="Entity ID of the resource")
|
||||
file_path: str = Field(..., description="File path of the resource")
|
||||
checksum: str = Field(..., description="File content checksum")
|
||||
size: int = Field(..., description="File size in bytes")
|
||||
created_at: float = Field(..., description="Creation timestamp")
|
||||
modified_at: float = Field(..., description="Modification timestamp")
|
||||
@@ -4,11 +4,13 @@ from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
from basic_memory.repository.entity_repository import EntityRepository
|
||||
from basic_memory.repository.observation_repository import ObservationRepository
|
||||
from basic_memory.repository.postgres_search_repository import PostgresSearchRepository
|
||||
from basic_memory.repository.search_repository import SearchRepository, SearchIndexRow
|
||||
from basic_memory.schemas.memory import MemoryUrl, memory_url_path
|
||||
from basic_memory.schemas.search import SearchItemType
|
||||
@@ -252,9 +254,6 @@ class ContextService:
|
||||
# Build the VALUES clause for entity IDs
|
||||
entity_id_values = ", ".join([str(i) for i in entity_ids])
|
||||
|
||||
# For compatibility with the old query, we still need this for filtering
|
||||
values = ", ".join([f"('{t}', {i})" for t, i in type_id_pairs])
|
||||
|
||||
# Parameters for bindings - include project_id for security filtering
|
||||
params = {
|
||||
"max_depth": max_depth,
|
||||
@@ -264,7 +263,14 @@ class ContextService:
|
||||
|
||||
# Build date and timeframe filters conditionally based on since parameter
|
||||
if since:
|
||||
params["since_date"] = since.isoformat() # pyright: ignore
|
||||
# SQLite accepts ISO strings, but Postgres/asyncpg requires datetime objects
|
||||
if isinstance(self.search_repository, PostgresSearchRepository):
|
||||
# asyncpg expects timezone-NAIVE datetime in UTC for DateTime(timezone=True) columns
|
||||
# even though the column stores timezone-aware values
|
||||
since_utc = since.astimezone(timezone.utc) if since.tzinfo else since
|
||||
params["since_date"] = since_utc.replace(tzinfo=None) # pyright: ignore
|
||||
else:
|
||||
params["since_date"] = since.isoformat() # pyright: ignore
|
||||
date_filter = "AND e.created_at >= :since_date"
|
||||
relation_date_filter = "AND e_from.created_at >= :since_date"
|
||||
timeframe_condition = "AND eg.relation_date >= :since_date"
|
||||
@@ -279,13 +285,210 @@ class ContextService:
|
||||
|
||||
# Use a CTE that operates directly on entity and relation tables
|
||||
# This avoids the overhead of the search_index virtual table
|
||||
query = text(f"""
|
||||
# Note: Postgres and SQLite have different CTE limitations:
|
||||
# - Postgres: doesn't allow multiple UNION ALL branches referencing the CTE
|
||||
# - SQLite: doesn't support LATERAL joins
|
||||
# So we need different queries for each database backend
|
||||
|
||||
# Detect database backend
|
||||
is_postgres = isinstance(self.search_repository, PostgresSearchRepository)
|
||||
|
||||
if is_postgres:
|
||||
query = self._build_postgres_query(
|
||||
entity_id_values,
|
||||
date_filter,
|
||||
project_filter,
|
||||
relation_date_filter,
|
||||
relation_project_filter,
|
||||
timeframe_condition,
|
||||
)
|
||||
else:
|
||||
# SQLite needs VALUES clause for exclusion (not needed for Postgres)
|
||||
values = ", ".join([f"('{t}', {i})" for t, i in type_id_pairs])
|
||||
query = self._build_sqlite_query(
|
||||
entity_id_values,
|
||||
date_filter,
|
||||
project_filter,
|
||||
relation_date_filter,
|
||||
relation_project_filter,
|
||||
timeframe_condition,
|
||||
values,
|
||||
)
|
||||
|
||||
result = await self.search_repository.execute_query(query, params=params)
|
||||
rows = result.all()
|
||||
|
||||
context_rows = [
|
||||
ContextResultRow(
|
||||
type=row.type,
|
||||
id=row.id,
|
||||
title=row.title,
|
||||
permalink=row.permalink,
|
||||
file_path=row.file_path,
|
||||
from_id=row.from_id,
|
||||
to_id=row.to_id,
|
||||
relation_type=row.relation_type,
|
||||
content=row.content,
|
||||
category=row.category,
|
||||
entity_id=row.entity_id,
|
||||
depth=row.depth,
|
||||
root_id=row.root_id,
|
||||
created_at=row.created_at,
|
||||
)
|
||||
for row in rows
|
||||
]
|
||||
return context_rows
|
||||
|
||||
def _build_postgres_query(
|
||||
self,
|
||||
entity_id_values: str,
|
||||
date_filter: str,
|
||||
project_filter: str,
|
||||
relation_date_filter: str,
|
||||
relation_project_filter: str,
|
||||
timeframe_condition: str,
|
||||
):
|
||||
"""Build Postgres-specific CTE query using LATERAL joins."""
|
||||
return text(f"""
|
||||
WITH RECURSIVE entity_graph AS (
|
||||
-- Base case: seed entities
|
||||
SELECT
|
||||
SELECT
|
||||
e.id,
|
||||
'entity' as type,
|
||||
e.title,
|
||||
e.title,
|
||||
e.permalink,
|
||||
e.file_path,
|
||||
CAST(NULL AS INTEGER) as from_id,
|
||||
CAST(NULL AS INTEGER) as to_id,
|
||||
CAST(NULL AS TEXT) as relation_type,
|
||||
CAST(NULL AS TEXT) as content,
|
||||
CAST(NULL AS TEXT) as category,
|
||||
CAST(NULL AS INTEGER) as entity_id,
|
||||
0 as depth,
|
||||
e.id as root_id,
|
||||
e.created_at,
|
||||
e.created_at as relation_date
|
||||
FROM entity e
|
||||
WHERE e.id IN ({entity_id_values})
|
||||
{date_filter}
|
||||
{project_filter}
|
||||
|
||||
UNION ALL
|
||||
|
||||
-- Fetch BOTH relations AND connected entities in a single recursive step
|
||||
-- Postgres only allows ONE reference to the recursive CTE in the recursive term
|
||||
-- We use CROSS JOIN LATERAL to generate two rows (relation + entity) from each traversal
|
||||
SELECT
|
||||
CASE
|
||||
WHEN step_type = 1 THEN r.id
|
||||
ELSE e.id
|
||||
END as id,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN 'relation'
|
||||
ELSE 'entity'
|
||||
END as type,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN r.relation_type || ': ' || r.to_name
|
||||
ELSE e.title
|
||||
END as title,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN ''
|
||||
ELSE COALESCE(e.permalink, '')
|
||||
END as permalink,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN e_from.file_path
|
||||
ELSE e.file_path
|
||||
END as file_path,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN r.from_id
|
||||
ELSE NULL
|
||||
END as from_id,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN r.to_id
|
||||
ELSE NULL
|
||||
END as to_id,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN r.relation_type
|
||||
ELSE NULL
|
||||
END as relation_type,
|
||||
CAST(NULL AS TEXT) as content,
|
||||
CAST(NULL AS TEXT) as category,
|
||||
CAST(NULL AS INTEGER) as entity_id,
|
||||
eg.depth + step_type as depth,
|
||||
eg.root_id,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN e_from.created_at
|
||||
ELSE e.created_at
|
||||
END as created_at,
|
||||
CASE
|
||||
WHEN step_type = 1 THEN e_from.created_at
|
||||
ELSE eg.relation_date
|
||||
END as relation_date
|
||||
FROM entity_graph eg
|
||||
CROSS JOIN LATERAL (VALUES (1), (2)) AS steps(step_type)
|
||||
JOIN relation r ON (
|
||||
eg.type = 'entity' AND
|
||||
(r.from_id = eg.id OR r.to_id = eg.id)
|
||||
)
|
||||
JOIN entity e_from ON (
|
||||
r.from_id = e_from.id
|
||||
{relation_project_filter}
|
||||
)
|
||||
LEFT JOIN entity e ON (
|
||||
step_type = 2 AND
|
||||
e.id = CASE
|
||||
WHEN r.from_id = eg.id THEN r.to_id
|
||||
ELSE r.from_id
|
||||
END
|
||||
{date_filter}
|
||||
{project_filter}
|
||||
)
|
||||
WHERE eg.depth < :max_depth
|
||||
AND (step_type = 1 OR (step_type = 2 AND e.id IS NOT NULL AND e.id != eg.id))
|
||||
{timeframe_condition}
|
||||
)
|
||||
-- Materialize and filter
|
||||
SELECT DISTINCT
|
||||
type,
|
||||
id,
|
||||
title,
|
||||
permalink,
|
||||
file_path,
|
||||
from_id,
|
||||
to_id,
|
||||
relation_type,
|
||||
content,
|
||||
category,
|
||||
entity_id,
|
||||
MIN(depth) as depth,
|
||||
root_id,
|
||||
created_at
|
||||
FROM entity_graph
|
||||
WHERE depth > 0
|
||||
GROUP BY type, id, title, permalink, file_path, from_id, to_id,
|
||||
relation_type, content, category, entity_id, root_id, created_at
|
||||
ORDER BY depth, type, id
|
||||
LIMIT :max_results
|
||||
""")
|
||||
|
||||
def _build_sqlite_query(
|
||||
self,
|
||||
entity_id_values: str,
|
||||
date_filter: str,
|
||||
project_filter: str,
|
||||
relation_date_filter: str,
|
||||
relation_project_filter: str,
|
||||
timeframe_condition: str,
|
||||
values: str,
|
||||
):
|
||||
"""Build SQLite-specific CTE query using multiple UNION ALL branches."""
|
||||
return text(f"""
|
||||
WITH RECURSIVE entity_graph AS (
|
||||
-- Base case: seed entities
|
||||
SELECT
|
||||
e.id,
|
||||
'entity' as type,
|
||||
e.title,
|
||||
e.permalink,
|
||||
e.file_path,
|
||||
NULL as from_id,
|
||||
@@ -311,7 +514,6 @@ class ContextService:
|
||||
r.id,
|
||||
'relation' as type,
|
||||
r.relation_type || ': ' || r.to_name as title,
|
||||
-- Relation model doesn't have permalink column - we'll generate it at runtime
|
||||
'' as permalink,
|
||||
e_from.file_path,
|
||||
r.from_id,
|
||||
@@ -322,7 +524,7 @@ class ContextService:
|
||||
NULL as entity_id,
|
||||
eg.depth + 1,
|
||||
eg.root_id,
|
||||
e_from.created_at, -- Use the from_entity's created_at since relation has no timestamp
|
||||
e_from.created_at,
|
||||
e_from.created_at as relation_date,
|
||||
CASE WHEN r.from_id = eg.id THEN 0 ELSE 1 END as is_incoming
|
||||
FROM entity_graph eg
|
||||
@@ -337,7 +539,6 @@ class ContextService:
|
||||
)
|
||||
LEFT JOIN entity e_to ON (r.to_id = e_to.id)
|
||||
WHERE eg.depth < :max_depth
|
||||
-- Ensure to_entity (if exists) also belongs to same project
|
||||
AND (r.to_id IS NULL OR e_to.project_id = :project_id)
|
||||
|
||||
UNION ALL
|
||||
@@ -347,9 +548,9 @@ class ContextService:
|
||||
e.id,
|
||||
'entity' as type,
|
||||
e.title,
|
||||
CASE
|
||||
WHEN e.permalink IS NULL THEN ''
|
||||
ELSE e.permalink
|
||||
CASE
|
||||
WHEN e.permalink IS NULL THEN ''
|
||||
ELSE e.permalink
|
||||
END as permalink,
|
||||
e.file_path,
|
||||
NULL as from_id,
|
||||
@@ -366,7 +567,7 @@ class ContextService:
|
||||
FROM entity_graph eg
|
||||
JOIN entity e ON (
|
||||
eg.type = 'relation' AND
|
||||
e.id = CASE
|
||||
e.id = CASE
|
||||
WHEN eg.is_incoming = 0 THEN eg.to_id
|
||||
ELSE eg.from_id
|
||||
END
|
||||
@@ -374,10 +575,9 @@ class ContextService:
|
||||
{project_filter}
|
||||
)
|
||||
WHERE eg.depth < :max_depth
|
||||
-- Only include entities connected by relations within timeframe if specified
|
||||
{timeframe_condition}
|
||||
)
|
||||
SELECT DISTINCT
|
||||
SELECT DISTINCT
|
||||
type,
|
||||
id,
|
||||
title,
|
||||
@@ -393,33 +593,9 @@ class ContextService:
|
||||
root_id,
|
||||
created_at
|
||||
FROM entity_graph
|
||||
WHERE (type, id) NOT IN ({values})
|
||||
GROUP BY
|
||||
type, id
|
||||
WHERE depth > 0
|
||||
GROUP BY type, id, title, permalink, file_path, from_id, to_id,
|
||||
relation_type, content, category, entity_id, root_id, created_at
|
||||
ORDER BY depth, type, id
|
||||
LIMIT :max_results
|
||||
""")
|
||||
|
||||
result = await self.search_repository.execute_query(query, params=params)
|
||||
rows = result.all()
|
||||
|
||||
context_rows = [
|
||||
ContextResultRow(
|
||||
type=row.type,
|
||||
id=row.id,
|
||||
title=row.title,
|
||||
permalink=row.permalink,
|
||||
file_path=row.file_path,
|
||||
from_id=row.from_id,
|
||||
to_id=row.to_id,
|
||||
relation_type=row.relation_type,
|
||||
content=row.content,
|
||||
category=row.category,
|
||||
entity_id=row.entity_id,
|
||||
depth=row.depth,
|
||||
root_id=row.root_id,
|
||||
created_at=row.created_at,
|
||||
)
|
||||
for row in rows
|
||||
]
|
||||
return context_rows
|
||||
|
||||
@@ -3,8 +3,10 @@
|
||||
import fnmatch
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Optional, Sequence
|
||||
|
||||
|
||||
from basic_memory.models import Entity
|
||||
from basic_memory.repository import EntityRepository
|
||||
from basic_memory.schemas.directory import DirectoryNode
|
||||
@@ -12,6 +14,17 @@ from basic_memory.schemas.directory import DirectoryNode
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: Entity) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
Returns the file's actual modification time, falling back to updated_at
|
||||
if mtime is not available.
|
||||
"""
|
||||
if entity.mtime:
|
||||
return datetime.fromtimestamp(entity.mtime).astimezone()
|
||||
return entity.updated_at
|
||||
|
||||
|
||||
class DirectoryService:
|
||||
"""Service for working with directory trees."""
|
||||
|
||||
@@ -77,7 +90,7 @@ class DirectoryService:
|
||||
entity_id=file.id,
|
||||
entity_type=file.entity_type,
|
||||
content_type=file.content_type,
|
||||
updated_at=file.updated_at,
|
||||
updated_at=_mtime_to_datetime(file),
|
||||
)
|
||||
|
||||
# Add to parent directory's children
|
||||
@@ -241,7 +254,7 @@ class DirectoryService:
|
||||
entity_id=file.id,
|
||||
entity_type=file.entity_type,
|
||||
content_type=file.content_type,
|
||||
updated_at=file.updated_at,
|
||||
updated_at=_mtime_to_datetime(file),
|
||||
)
|
||||
|
||||
# Add to parent directory's children
|
||||
|
||||
@@ -8,6 +8,7 @@ import yaml
|
||||
from loguru import logger
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
|
||||
|
||||
from basic_memory.config import ProjectConfig, BasicMemoryConfig
|
||||
from basic_memory.file_utils import (
|
||||
has_frontmatter,
|
||||
@@ -106,6 +107,9 @@ class EntityService(BaseService[EntityModel]):
|
||||
4. Generate new unique permalink from file path
|
||||
|
||||
Enhanced to detect and handle character-related conflicts.
|
||||
|
||||
Note: Uses lightweight repository methods that skip eager loading of
|
||||
observations and relations for better performance during bulk operations.
|
||||
"""
|
||||
file_path_str = Path(file_path).as_posix()
|
||||
|
||||
@@ -122,16 +126,20 @@ class EntityService(BaseService[EntityModel]):
|
||||
# If markdown has explicit permalink, try to validate it
|
||||
if markdown and markdown.frontmatter.permalink:
|
||||
desired_permalink = markdown.frontmatter.permalink
|
||||
existing = await self.repository.get_by_permalink(desired_permalink)
|
||||
# Use lightweight method - we only need to check file_path
|
||||
existing_file_path = await self.repository.get_file_path_for_permalink(
|
||||
desired_permalink
|
||||
)
|
||||
|
||||
# If no conflict or it's our own file, use as is
|
||||
if not existing or existing.file_path == file_path_str:
|
||||
if not existing_file_path or existing_file_path == file_path_str:
|
||||
return desired_permalink
|
||||
|
||||
# For existing files, try to find current permalink
|
||||
existing = await self.repository.get_by_file_path(file_path_str)
|
||||
if existing:
|
||||
return existing.permalink
|
||||
# Use lightweight method - we only need the permalink
|
||||
existing_permalink = await self.repository.get_permalink_for_file_path(file_path_str)
|
||||
if existing_permalink:
|
||||
return existing_permalink
|
||||
|
||||
# New file - generate permalink
|
||||
if markdown and markdown.frontmatter.permalink:
|
||||
@@ -140,9 +148,10 @@ class EntityService(BaseService[EntityModel]):
|
||||
desired_permalink = generate_permalink(file_path_str)
|
||||
|
||||
# Make unique if needed - enhanced to handle character conflicts
|
||||
# Use lightweight existence check instead of loading full entity
|
||||
permalink = desired_permalink
|
||||
suffix = 1
|
||||
while await self.repository.get_by_permalink(permalink):
|
||||
while await self.repository.permalink_exists(permalink):
|
||||
permalink = f"{desired_permalink}-{suffix}"
|
||||
suffix += 1
|
||||
logger.debug(f"creating unique permalink: {permalink}")
|
||||
@@ -224,8 +233,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
final_content = dump_frontmatter(post)
|
||||
checksum = await self.file_service.write_file(file_path, final_content)
|
||||
|
||||
# parse entity from file
|
||||
entity_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# parse entity from content we just wrote (avoids re-reading file for cloud compatibility)
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=final_content,
|
||||
)
|
||||
|
||||
# create entity
|
||||
created = await self.create_entity_from_markdown(file_path, entity_markdown)
|
||||
@@ -245,8 +257,12 @@ class EntityService(BaseService[EntityModel]):
|
||||
# Convert file path string to Path
|
||||
file_path = Path(entity.file_path)
|
||||
|
||||
# Read existing frontmatter from the file if it exists
|
||||
existing_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# Read existing content via file_service (for cloud compatibility)
|
||||
existing_content = await self.file_service.read_file_content(file_path)
|
||||
existing_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=existing_content,
|
||||
)
|
||||
|
||||
# Parse content frontmatter to check for user-specified permalink and entity_type
|
||||
content_markdown = None
|
||||
@@ -302,8 +318,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
final_content = dump_frontmatter(merged_post)
|
||||
checksum = await self.file_service.write_file(file_path, final_content)
|
||||
|
||||
# parse entity from file
|
||||
entity_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# parse entity from content we just wrote (avoids re-reading file for cloud compatibility)
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=final_content,
|
||||
)
|
||||
|
||||
# update entity in db
|
||||
entity = await self.update_entity_and_observations(file_path, entity_markdown)
|
||||
@@ -378,7 +397,9 @@ class EntityService(BaseService[EntityModel]):
|
||||
Uses UPSERT approach to handle permalink/file_path conflicts cleanly.
|
||||
"""
|
||||
logger.debug(f"Creating entity: {markdown.frontmatter.title} file_path: {file_path}")
|
||||
model = entity_model_from_markdown(file_path, markdown)
|
||||
model = entity_model_from_markdown(
|
||||
file_path, markdown, project_id=self.repository.project_id
|
||||
)
|
||||
|
||||
# Mark as incomplete because we still need to add relations
|
||||
model.checksum = None
|
||||
@@ -408,6 +429,7 @@ class EntityService(BaseService[EntityModel]):
|
||||
# add new observations
|
||||
observations = [
|
||||
Observation(
|
||||
project_id=self.observation_repository.project_id,
|
||||
entity_id=db_entity.id,
|
||||
content=obs.content,
|
||||
category=obs.category,
|
||||
@@ -448,8 +470,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
import asyncio
|
||||
|
||||
# Create tasks for all relation lookups
|
||||
# Use strict=True to disable fuzzy search - only exact matches should create resolved relations
|
||||
# This ensures forward references (links to non-existent entities) remain unresolved (to_id=NULL)
|
||||
lookup_tasks = [
|
||||
self.link_resolver.resolve_link(rel.target) for rel in markdown.relations
|
||||
self.link_resolver.resolve_link(rel.target, strict=True)
|
||||
for rel in markdown.relations
|
||||
]
|
||||
|
||||
# Execute all lookups in parallel
|
||||
@@ -471,6 +496,7 @@ class EntityService(BaseService[EntityModel]):
|
||||
|
||||
# Create the relation
|
||||
relation = Relation(
|
||||
project_id=self.relation_repository.project_id,
|
||||
from_id=db_entity.id,
|
||||
to_id=target_id,
|
||||
to_name=target_name,
|
||||
@@ -543,8 +569,11 @@ class EntityService(BaseService[EntityModel]):
|
||||
# Write the updated content back to the file
|
||||
checksum = await self.file_service.write_file(file_path, new_content)
|
||||
|
||||
# Parse the updated file to get new observations/relations
|
||||
entity_markdown = await self.entity_parser.parse_file(file_path)
|
||||
# Parse the content we just wrote (avoids re-reading file for cloud compatibility)
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=file_path,
|
||||
content=new_content,
|
||||
)
|
||||
|
||||
# Update entity and its relationships
|
||||
entity = await self.update_entity_and_observations(file_path, entity_markdown)
|
||||
@@ -763,23 +792,20 @@ class EntityService(BaseService[EntityModel]):
|
||||
raise ValueError(f"Invalid destination path: {destination_path}")
|
||||
|
||||
# 3. Validate paths
|
||||
source_file = project_config.home / current_path
|
||||
destination_file = project_config.home / destination_path
|
||||
|
||||
# Validate source exists
|
||||
if not source_file.exists():
|
||||
# NOTE: In tenantless/cloud mode, we cannot rely on local filesystem paths.
|
||||
# Use FileService for existence checks and moving.
|
||||
if not await self.file_service.exists(current_path):
|
||||
raise ValueError(f"Source file not found: {current_path}")
|
||||
|
||||
# Check if destination already exists
|
||||
if destination_file.exists():
|
||||
if await self.file_service.exists(destination_path):
|
||||
raise ValueError(f"Destination already exists: {destination_path}")
|
||||
|
||||
try:
|
||||
# 4. Create destination directory if needed
|
||||
destination_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
# 4. Ensure destination directory if needed (no-op for S3)
|
||||
await self.file_service.ensure_directory(Path(destination_path).parent)
|
||||
|
||||
# 5. Move physical file
|
||||
source_file.rename(destination_file)
|
||||
# 5. Move physical file via FileService (filesystem rename or cloud move)
|
||||
await self.file_service.move_file(current_path, destination_path)
|
||||
logger.info(f"Moved file: {current_path} -> {destination_path}")
|
||||
|
||||
# 6. Prepare database updates
|
||||
@@ -818,12 +844,14 @@ class EntityService(BaseService[EntityModel]):
|
||||
|
||||
except Exception as e:
|
||||
# Rollback: try to restore original file location if move succeeded
|
||||
if destination_file.exists() and not source_file.exists():
|
||||
try:
|
||||
destination_file.rename(source_file)
|
||||
try:
|
||||
if await self.file_service.exists(
|
||||
destination_path
|
||||
) and not await self.file_service.exists(current_path):
|
||||
await self.file_service.move_file(destination_path, current_path)
|
||||
logger.info(f"Rolled back file move: {destination_path} -> {current_path}")
|
||||
except Exception as rollback_error: # pragma: no cover
|
||||
logger.error(f"Failed to rollback file move: {rollback_error}")
|
||||
except Exception as rollback_error: # pragma: no cover
|
||||
logger.error(f"Failed to rollback file move: {rollback_error}")
|
||||
|
||||
# Re-raise the original error with context
|
||||
raise ValueError(f"Move failed: {str(e)}") from e
|
||||
|
||||
@@ -3,15 +3,16 @@
|
||||
import asyncio
|
||||
import hashlib
|
||||
import mimetypes
|
||||
from os import stat_result
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Tuple, Union
|
||||
|
||||
import aiofiles
|
||||
|
||||
import yaml
|
||||
|
||||
from basic_memory import file_utils
|
||||
from basic_memory.file_utils import FileError, ParseError
|
||||
from basic_memory.file_utils import FileError, FileMetadata, ParseError
|
||||
from basic_memory.markdown.markdown_processor import MarkdownProcessor
|
||||
from basic_memory.models import Entity as EntityModel
|
||||
from basic_memory.schemas import Entity as EntitySchema
|
||||
@@ -220,6 +221,41 @@ class FileService:
|
||||
logger.exception("File read error", path=str(full_path), error=str(e))
|
||||
raise FileOperationError(f"Failed to read file: {e}")
|
||||
|
||||
async def read_file_bytes(self, path: FilePath) -> bytes:
|
||||
"""Read file content as bytes using true async I/O with aiofiles.
|
||||
|
||||
This method reads files in binary mode, suitable for non-text files
|
||||
like images, PDFs, etc. For cloud compatibility with S3FileService.
|
||||
|
||||
Args:
|
||||
path: Path to read (Path or string)
|
||||
|
||||
Returns:
|
||||
File content as bytes
|
||||
|
||||
Raises:
|
||||
FileOperationError: If read fails
|
||||
"""
|
||||
# Convert string to Path if needed
|
||||
path_obj = self.base_path / path if isinstance(path, str) else path
|
||||
full_path = path_obj if path_obj.is_absolute() else self.base_path / path_obj
|
||||
|
||||
try:
|
||||
logger.debug("Reading file bytes", operation="read_file_bytes", path=str(full_path))
|
||||
async with aiofiles.open(full_path, mode="rb") as f:
|
||||
content = await f.read()
|
||||
|
||||
logger.debug(
|
||||
"File read completed",
|
||||
path=str(full_path),
|
||||
content_length=len(content),
|
||||
)
|
||||
return content
|
||||
|
||||
except Exception as e:
|
||||
logger.exception("File read error", path=str(full_path), error=str(e))
|
||||
raise FileOperationError(f"Failed to read file: {e}")
|
||||
|
||||
async def read_file(self, path: FilePath) -> Tuple[str, str]:
|
||||
"""Read file and compute checksum using true async I/O.
|
||||
|
||||
@@ -276,6 +312,43 @@ class FileService:
|
||||
full_path = path_obj if path_obj.is_absolute() else self.base_path / path_obj
|
||||
full_path.unlink(missing_ok=True)
|
||||
|
||||
async def move_file(self, source: FilePath, destination: FilePath) -> None:
|
||||
"""Move/rename a file from source to destination.
|
||||
|
||||
This method abstracts the underlying storage (filesystem vs cloud).
|
||||
Default implementation uses atomic filesystem rename, but cloud-backed
|
||||
implementations (e.g., S3) can override to copy+delete.
|
||||
|
||||
Args:
|
||||
source: Source path (relative to base_path or absolute)
|
||||
destination: Destination path (relative to base_path or absolute)
|
||||
|
||||
Raises:
|
||||
FileOperationError: If the move fails
|
||||
"""
|
||||
# Convert strings to Paths and resolve relative paths against base_path
|
||||
src_obj = self.base_path / source if isinstance(source, str) else source
|
||||
dst_obj = self.base_path / destination if isinstance(destination, str) else destination
|
||||
src_full = src_obj if src_obj.is_absolute() else self.base_path / src_obj
|
||||
dst_full = dst_obj if dst_obj.is_absolute() else self.base_path / dst_obj
|
||||
|
||||
try:
|
||||
# Ensure destination directory exists
|
||||
await self.ensure_directory(dst_full.parent)
|
||||
|
||||
# Use semaphore for concurrency control and run blocking rename in executor
|
||||
async with self._file_semaphore:
|
||||
loop = asyncio.get_event_loop()
|
||||
await loop.run_in_executor(None, lambda: src_full.rename(dst_full))
|
||||
except Exception as e:
|
||||
logger.exception(
|
||||
"File move error",
|
||||
source=str(src_full),
|
||||
destination=str(dst_full),
|
||||
error=str(e),
|
||||
)
|
||||
raise FileOperationError(f"Failed to move file {source} -> {destination}: {e}")
|
||||
|
||||
async def update_frontmatter(self, path: FilePath, updates: Dict[str, Any]) -> str:
|
||||
"""Update frontmatter fields in a file while preserving all content.
|
||||
|
||||
@@ -381,20 +454,31 @@ class FileService:
|
||||
logger.error("Failed to compute checksum", path=str(full_path), error=str(e))
|
||||
raise FileError(f"Failed to compute checksum for {path}: {e}")
|
||||
|
||||
def file_stats(self, path: FilePath) -> stat_result:
|
||||
"""Return file stats for a given path.
|
||||
async def get_file_metadata(self, path: FilePath) -> FileMetadata:
|
||||
"""Return file metadata for a given path.
|
||||
|
||||
This method is async to support cloud implementations (S3FileService)
|
||||
where file metadata requires async operations (head_object).
|
||||
|
||||
Args:
|
||||
path: Path to the file (Path or string)
|
||||
|
||||
Returns:
|
||||
File statistics
|
||||
FileMetadata with size, created_at, and modified_at
|
||||
"""
|
||||
# Convert string to Path if needed
|
||||
path_obj = self.base_path / path if isinstance(path, str) else path
|
||||
full_path = path_obj if path_obj.is_absolute() else self.base_path / path_obj
|
||||
# get file timestamps
|
||||
return full_path.stat()
|
||||
|
||||
# Run blocking stat() in thread pool to maintain async compatibility
|
||||
loop = asyncio.get_event_loop()
|
||||
stat_result = await loop.run_in_executor(None, full_path.stat)
|
||||
|
||||
return FileMetadata(
|
||||
size=stat_result.st_size,
|
||||
created_at=datetime.fromtimestamp(stat_result.st_ctime).astimezone(),
|
||||
modified_at=datetime.fromtimestamp(stat_result.st_mtime).astimezone(),
|
||||
)
|
||||
|
||||
def content_type(self, path: FilePath) -> str:
|
||||
"""Return content_type for a given path.
|
||||
|
||||
@@ -5,8 +5,10 @@ to ensure consistent application startup across all entry points.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory import db
|
||||
@@ -104,6 +106,12 @@ async def initialize_file_sync(
|
||||
# Get active projects
|
||||
active_projects = await project_repository.get_active_projects()
|
||||
|
||||
# Filter to constrained project if MCP server was started with --project
|
||||
constrained_project = os.environ.get("BASIC_MEMORY_MCP_PROJECT")
|
||||
if constrained_project:
|
||||
active_projects = [p for p in active_projects if p.name == constrained_project]
|
||||
logger.info(f"Background sync constrained to project: {constrained_project}")
|
||||
|
||||
# Start sync for all projects as background tasks (non-blocking)
|
||||
async def sync_project_background(project: Project):
|
||||
"""Sync a single project in the background."""
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from typing import Optional, Tuple
|
||||
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from basic_memory.models import Entity
|
||||
|
||||
@@ -8,6 +8,7 @@ from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional, Sequence
|
||||
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy import text
|
||||
|
||||
@@ -23,9 +24,6 @@ from basic_memory.config import WATCH_STATUS_JSON, ConfigManager, get_project_co
|
||||
from basic_memory.utils import generate_permalink
|
||||
|
||||
|
||||
config = ConfigManager().config
|
||||
|
||||
|
||||
class ProjectService:
|
||||
"""Service for managing Basic Memory projects."""
|
||||
|
||||
@@ -143,6 +141,7 @@ class ProjectService:
|
||||
"""
|
||||
# If project_root is set, constrain all projects to that directory
|
||||
project_root = self.config_manager.config.project_root
|
||||
sanitized_name = None
|
||||
if project_root:
|
||||
base_path = Path(project_root)
|
||||
|
||||
@@ -199,14 +198,15 @@ class ProjectService:
|
||||
f"Projects cannot share directory trees."
|
||||
)
|
||||
|
||||
# First add to config file (this will validate the project doesn't exist)
|
||||
project_config = self.config_manager.add_project(name, resolved_path)
|
||||
if not self.config_manager.config.cloud_mode:
|
||||
# First add to config file (this will validate the project doesn't exist)
|
||||
self.config_manager.add_project(name, resolved_path)
|
||||
|
||||
# Then add to database
|
||||
project_data = {
|
||||
"name": name,
|
||||
"path": resolved_path,
|
||||
"permalink": generate_permalink(project_config.name),
|
||||
"permalink": sanitized_name,
|
||||
"is_active": True,
|
||||
# Don't set is_default=False to avoid UNIQUE constraint issues
|
||||
# Let it default to NULL, only set to True when explicitly making default
|
||||
@@ -766,25 +766,42 @@ class ProjectService:
|
||||
)
|
||||
|
||||
# Query for monthly entity creation (project filtered)
|
||||
# Use different date formatting for SQLite vs Postgres
|
||||
from basic_memory.config import DatabaseBackend
|
||||
|
||||
is_postgres = self.config_manager.config.database_backend == DatabaseBackend.POSTGRES
|
||||
date_format = (
|
||||
"to_char(created_at, 'YYYY-MM')" if is_postgres else "strftime('%Y-%m', created_at)"
|
||||
)
|
||||
|
||||
# Postgres needs datetime objects, SQLite needs ISO strings
|
||||
six_months_param = six_months_ago if is_postgres else six_months_ago.isoformat()
|
||||
|
||||
entity_growth_result = await self.repository.execute_query(
|
||||
text("""
|
||||
SELECT
|
||||
strftime('%Y-%m', created_at) AS month,
|
||||
text(f"""
|
||||
SELECT
|
||||
{date_format} AS month,
|
||||
COUNT(*) AS count
|
||||
FROM entity
|
||||
WHERE created_at >= :six_months_ago AND project_id = :project_id
|
||||
GROUP BY month
|
||||
ORDER BY month
|
||||
"""),
|
||||
{"six_months_ago": six_months_ago.isoformat(), "project_id": project_id},
|
||||
{"six_months_ago": six_months_param, "project_id": project_id},
|
||||
)
|
||||
entity_growth = {row[0]: row[1] for row in entity_growth_result.fetchall()}
|
||||
|
||||
# Query for monthly observation creation (project filtered)
|
||||
date_format_entity = (
|
||||
"to_char(entity.created_at, 'YYYY-MM')"
|
||||
if is_postgres
|
||||
else "strftime('%Y-%m', entity.created_at)"
|
||||
)
|
||||
|
||||
observation_growth_result = await self.repository.execute_query(
|
||||
text("""
|
||||
SELECT
|
||||
strftime('%Y-%m', entity.created_at) AS month,
|
||||
text(f"""
|
||||
SELECT
|
||||
{date_format_entity} AS month,
|
||||
COUNT(*) AS count
|
||||
FROM observation
|
||||
INNER JOIN entity ON observation.entity_id = entity.id
|
||||
@@ -792,15 +809,15 @@ class ProjectService:
|
||||
GROUP BY month
|
||||
ORDER BY month
|
||||
"""),
|
||||
{"six_months_ago": six_months_ago.isoformat(), "project_id": project_id},
|
||||
{"six_months_ago": six_months_param, "project_id": project_id},
|
||||
)
|
||||
observation_growth = {row[0]: row[1] for row in observation_growth_result.fetchall()}
|
||||
|
||||
# Query for monthly relation creation (project filtered)
|
||||
relation_growth_result = await self.repository.execute_query(
|
||||
text("""
|
||||
SELECT
|
||||
strftime('%Y-%m', entity.created_at) AS month,
|
||||
text(f"""
|
||||
SELECT
|
||||
{date_format_entity} AS month,
|
||||
COUNT(*) AS count
|
||||
FROM relation
|
||||
INNER JOIN entity ON relation.from_id = entity.id
|
||||
@@ -808,7 +825,7 @@ class ProjectService:
|
||||
GROUP BY month
|
||||
ORDER BY month
|
||||
"""),
|
||||
{"six_months_ago": six_months_ago.isoformat(), "project_id": project_id},
|
||||
{"six_months_ago": six_months_param, "project_id": project_id},
|
||||
)
|
||||
relation_growth = {row[0]: row[1] for row in relation_growth_result.fetchall()}
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ import ast
|
||||
from datetime import datetime
|
||||
from typing import List, Optional, Set
|
||||
|
||||
|
||||
from dateparser import parse
|
||||
from fastapi import BackgroundTasks
|
||||
from loguru import logger
|
||||
@@ -15,6 +16,21 @@ from basic_memory.repository.search_repository import SearchRepository, SearchIn
|
||||
from basic_memory.schemas.search import SearchQuery, SearchItemType
|
||||
from basic_memory.services import FileService
|
||||
|
||||
# Maximum size for content_stems field to stay under Postgres's 8KB index row limit.
|
||||
# We use 6000 characters to leave headroom for other indexed columns and overhead.
|
||||
MAX_CONTENT_STEMS_SIZE = 6000
|
||||
|
||||
|
||||
def _mtime_to_datetime(entity: Entity) -> datetime:
|
||||
"""Convert entity mtime (file modification time) to datetime.
|
||||
|
||||
Returns the file's actual modification time, falling back to updated_at
|
||||
if mtime is not available.
|
||||
"""
|
||||
if entity.mtime:
|
||||
return datetime.fromtimestamp(entity.mtime).astimezone()
|
||||
return entity.updated_at
|
||||
|
||||
|
||||
class SearchService:
|
||||
"""Service for search operations.
|
||||
@@ -156,22 +172,24 @@ class SearchService:
|
||||
self,
|
||||
entity: Entity,
|
||||
background_tasks: Optional[BackgroundTasks] = None,
|
||||
content: str | None = None,
|
||||
) -> None:
|
||||
if background_tasks:
|
||||
background_tasks.add_task(self.index_entity_data, entity)
|
||||
background_tasks.add_task(self.index_entity_data, entity, content)
|
||||
else:
|
||||
await self.index_entity_data(entity)
|
||||
await self.index_entity_data(entity, content)
|
||||
|
||||
async def index_entity_data(
|
||||
self,
|
||||
entity: Entity,
|
||||
content: str | None = None,
|
||||
) -> None:
|
||||
# delete all search index data associated with entity
|
||||
await self.repository.delete_by_entity_id(entity_id=entity.id)
|
||||
|
||||
# reindex
|
||||
await self.index_entity_markdown(
|
||||
entity
|
||||
entity, content
|
||||
) if entity.is_markdown else await self.index_entity_file(entity)
|
||||
|
||||
async def index_entity_file(
|
||||
@@ -185,12 +203,13 @@ class SearchService:
|
||||
entity_id=entity.id,
|
||||
type=SearchItemType.ENTITY.value,
|
||||
title=entity.title,
|
||||
permalink=entity.permalink, # Required for Postgres NOT NULL constraint
|
||||
file_path=entity.file_path,
|
||||
metadata={
|
||||
"entity_type": entity.entity_type,
|
||||
},
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
@@ -198,9 +217,14 @@ class SearchService:
|
||||
async def index_entity_markdown(
|
||||
self,
|
||||
entity: Entity,
|
||||
content: str | None = None,
|
||||
) -> None:
|
||||
"""Index an entity and all its observations and relations.
|
||||
|
||||
Args:
|
||||
entity: The entity to index
|
||||
content: Optional pre-loaded content (avoids file read). If None, will read from file.
|
||||
|
||||
Indexing structure:
|
||||
1. Entities
|
||||
- permalink: direct from entity (e.g., "specs/search")
|
||||
@@ -229,7 +253,9 @@ class SearchService:
|
||||
title_variants = self._generate_variants(entity.title)
|
||||
content_stems.extend(title_variants)
|
||||
|
||||
content = await self.file_service.read_entity_content(entity)
|
||||
# Use provided content or read from file
|
||||
if content is None:
|
||||
content = await self.file_service.read_entity_content(entity)
|
||||
if content:
|
||||
content_stems.append(content)
|
||||
content_snippet = f"{content[:250]}"
|
||||
@@ -246,6 +272,10 @@ class SearchService:
|
||||
|
||||
entity_content_stems = "\n".join(p for p in content_stems if p and p.strip())
|
||||
|
||||
# Truncate to stay under Postgres's 8KB index row limit
|
||||
if len(entity_content_stems) > MAX_CONTENT_STEMS_SIZE:
|
||||
entity_content_stems = entity_content_stems[:MAX_CONTENT_STEMS_SIZE]
|
||||
|
||||
# Add entity row
|
||||
rows_to_index.append(
|
||||
SearchIndexRow(
|
||||
@@ -261,17 +291,28 @@ class SearchService:
|
||||
"entity_type": entity.entity_type,
|
||||
},
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
|
||||
# Add observation rows
|
||||
# Add observation rows - dedupe by permalink to avoid unique constraint violations
|
||||
# Two observations with same entity/category/content generate identical permalinks
|
||||
seen_permalinks: set[str] = {entity.permalink} if entity.permalink else set()
|
||||
for obs in entity.observations:
|
||||
obs_permalink = obs.permalink
|
||||
if obs_permalink in seen_permalinks:
|
||||
logger.debug(f"Skipping duplicate observation permalink: {obs_permalink}")
|
||||
continue
|
||||
seen_permalinks.add(obs_permalink)
|
||||
|
||||
# Index with parent entity's file path since that's where it's defined
|
||||
obs_content_stems = "\n".join(
|
||||
p for p in self._generate_variants(obs.content) if p and p.strip()
|
||||
)
|
||||
# Truncate to stay under Postgres's 8KB index row limit
|
||||
if len(obs_content_stems) > MAX_CONTENT_STEMS_SIZE:
|
||||
obs_content_stems = obs_content_stems[:MAX_CONTENT_STEMS_SIZE]
|
||||
rows_to_index.append(
|
||||
SearchIndexRow(
|
||||
id=obs.id,
|
||||
@@ -279,7 +320,7 @@ class SearchService:
|
||||
title=f"{obs.category}: {obs.content[:100]}...",
|
||||
content_stems=obs_content_stems,
|
||||
content_snippet=obs.content,
|
||||
permalink=obs.permalink,
|
||||
permalink=obs_permalink,
|
||||
file_path=entity.file_path,
|
||||
category=obs.category,
|
||||
entity_id=entity.id,
|
||||
@@ -287,7 +328,7 @@ class SearchService:
|
||||
"tags": obs.tags,
|
||||
},
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
@@ -317,7 +358,7 @@ class SearchService:
|
||||
to_id=rel.to_id,
|
||||
relation_type=rel.relation_type,
|
||||
created_at=entity.created_at,
|
||||
updated_at=entity.updated_at,
|
||||
updated_at=_mtime_to_datetime(entity),
|
||||
project_id=entity.project_id,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -10,7 +10,7 @@ from pathlib import Path
|
||||
from typing import AsyncIterator, Dict, List, Optional, Set, Tuple
|
||||
|
||||
import aiofiles.os
|
||||
import logfire
|
||||
|
||||
from loguru import logger
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
|
||||
@@ -26,7 +26,7 @@ from basic_memory.repository import (
|
||||
ObservationRepository,
|
||||
ProjectRepository,
|
||||
)
|
||||
from basic_memory.repository.search_repository import SearchRepository
|
||||
from basic_memory.repository.search_repository import create_search_repository
|
||||
from basic_memory.services import EntityService, FileService
|
||||
from basic_memory.services.exceptions import SyncFatalError
|
||||
from basic_memory.services.link_resolver import LinkResolver
|
||||
@@ -215,17 +215,12 @@ class SyncService:
|
||||
f"path={path}, error={error}"
|
||||
)
|
||||
|
||||
# Record metric for file failure
|
||||
logfire.metric_counter("sync.circuit_breaker.failures").add(1)
|
||||
|
||||
# Log when threshold is reached
|
||||
if failure_info.count >= MAX_CONSECUTIVE_FAILURES:
|
||||
logger.error(
|
||||
f"File {path} has failed {MAX_CONSECUTIVE_FAILURES} times and will be skipped. "
|
||||
f"First failure: {failure_info.first_failure}, Last error: {error}"
|
||||
)
|
||||
# Record metric for file being blocked by circuit breaker
|
||||
logfire.metric_counter("sync.circuit_breaker.blocked_files").add(1)
|
||||
else:
|
||||
# Create new failure record
|
||||
self._file_failures[path] = FileFailureInfo(
|
||||
@@ -255,7 +250,6 @@ class SyncService:
|
||||
logger.info(f"Clearing failure history for {path} after successful sync")
|
||||
del self._file_failures[path]
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync(
|
||||
self, directory: Path, project_name: Optional[str] = None, force_full: bool = False
|
||||
) -> SyncReport:
|
||||
@@ -282,63 +276,58 @@ class SyncService:
|
||||
)
|
||||
|
||||
# sync moves first
|
||||
with logfire.span("process_moves", move_count=len(report.moves)):
|
||||
for old_path, new_path in report.moves.items():
|
||||
# in the case where a file has been deleted and replaced by another file
|
||||
# it will show up in the move and modified lists, so handle it in modified
|
||||
if new_path in report.modified:
|
||||
report.modified.remove(new_path)
|
||||
logger.debug(
|
||||
f"File marked as moved and modified: old_path={old_path}, new_path={new_path}"
|
||||
)
|
||||
else:
|
||||
await self.handle_move(old_path, new_path)
|
||||
for old_path, new_path in report.moves.items():
|
||||
# in the case where a file has been deleted and replaced by another file
|
||||
# it will show up in the move and modified lists, so handle it in modified
|
||||
if new_path in report.modified:
|
||||
report.modified.remove(new_path)
|
||||
logger.debug(
|
||||
f"File marked as moved and modified: old_path={old_path}, new_path={new_path}"
|
||||
)
|
||||
else:
|
||||
await self.handle_move(old_path, new_path)
|
||||
|
||||
# deleted next
|
||||
with logfire.span("process_deletes", delete_count=len(report.deleted)):
|
||||
for path in report.deleted:
|
||||
await self.handle_delete(path)
|
||||
for path in report.deleted:
|
||||
await self.handle_delete(path)
|
||||
|
||||
# then new and modified
|
||||
with logfire.span("process_new_files", new_count=len(report.new)):
|
||||
for path in report.new:
|
||||
entity, _ = await self.sync_file(path, new=True)
|
||||
for path in report.new:
|
||||
entity, _ = await self.sync_file(path, new=True)
|
||||
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
)
|
||||
|
||||
with logfire.span("process_modified_files", modified_count=len(report.modified)):
|
||||
for path in report.modified:
|
||||
entity, _ = await self.sync_file(path, new=False)
|
||||
for path in report.modified:
|
||||
entity, _ = await self.sync_file(path, new=False)
|
||||
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
# Track if file was skipped
|
||||
if entity is None and await self._should_skip_file(path):
|
||||
failure_info = self._file_failures[path]
|
||||
report.skipped_files.append(
|
||||
SkippedFile(
|
||||
path=path,
|
||||
reason=failure_info.last_error,
|
||||
failure_count=failure_info.count,
|
||||
first_failed=failure_info.first_failure,
|
||||
)
|
||||
)
|
||||
|
||||
# Only resolve relations if there were actual changes
|
||||
# If no files changed, no new unresolved relations could have been created
|
||||
with logfire.span("resolve_relations"):
|
||||
if report.total > 0:
|
||||
await self.resolve_relations()
|
||||
else:
|
||||
logger.info("Skipping relation resolution - no file changes detected")
|
||||
if report.total > 0:
|
||||
await self.resolve_relations()
|
||||
else:
|
||||
logger.info("Skipping relation resolution - no file changes detected")
|
||||
|
||||
# Update scan watermark after successful sync
|
||||
# Use the timestamp from sync start (not end) to ensure we catch files
|
||||
@@ -361,15 +350,6 @@ class SyncService:
|
||||
|
||||
duration_ms = int((time.time() - start_time) * 1000)
|
||||
|
||||
# Record metrics for sync operation
|
||||
logfire.metric_histogram("sync.duration", unit="ms").record(duration_ms)
|
||||
logfire.metric_counter("sync.files.new").add(len(report.new))
|
||||
logfire.metric_counter("sync.files.modified").add(len(report.modified))
|
||||
logfire.metric_counter("sync.files.deleted").add(len(report.deleted))
|
||||
logfire.metric_counter("sync.files.moved").add(len(report.moves))
|
||||
if report.skipped_files:
|
||||
logfire.metric_counter("sync.files.skipped").add(len(report.skipped_files))
|
||||
|
||||
# Log summary with skipped files if any
|
||||
if report.skipped_files:
|
||||
logger.warning(
|
||||
@@ -390,7 +370,6 @@ class SyncService:
|
||||
|
||||
return report
|
||||
|
||||
@logfire.instrument()
|
||||
async def scan(self, directory, force_full: bool = False):
|
||||
"""Smart scan using watermark and file count for large project optimization.
|
||||
|
||||
@@ -472,12 +451,6 @@ class SyncService:
|
||||
logger.warning("No scan watermark available, falling back to full scan")
|
||||
file_paths_to_scan = await self._scan_directory_full(directory)
|
||||
|
||||
# Record scan type metric
|
||||
logfire.metric_counter(f"sync.scan.{scan_type}").add(1)
|
||||
logfire.metric_histogram("sync.scan.files_scanned", unit="files").record(
|
||||
len(file_paths_to_scan)
|
||||
)
|
||||
|
||||
# Step 3: Process each file with mtime-based comparison
|
||||
scanned_paths: Set[str] = set()
|
||||
changed_checksums: Dict[str, str] = {}
|
||||
@@ -589,7 +562,6 @@ class SyncService:
|
||||
report.checksums = changed_checksums
|
||||
|
||||
scan_duration_ms = int((time.time() - scan_start_time) * 1000)
|
||||
logfire.metric_histogram("sync.scan.duration", unit="ms").record(scan_duration_ms)
|
||||
|
||||
logger.info(
|
||||
f"Completed {scan_type} scan for directory {directory} in {scan_duration_ms}ms, "
|
||||
@@ -599,7 +571,6 @@ class SyncService:
|
||||
)
|
||||
return report
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync_file(
|
||||
self, path: str, new: bool = True
|
||||
) -> Tuple[Optional[Entity], Optional[str]]:
|
||||
@@ -654,7 +625,6 @@ class SyncService:
|
||||
|
||||
return None, None
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync_markdown_file(self, path: str, new: bool = True) -> Tuple[Optional[Entity], str]:
|
||||
"""Sync a markdown file with full processing.
|
||||
|
||||
@@ -672,12 +642,19 @@ class SyncService:
|
||||
file_contains_frontmatter = has_frontmatter(file_content)
|
||||
|
||||
# Get file timestamps for tracking modification times
|
||||
file_stats = self.file_service.file_stats(path)
|
||||
created = datetime.fromtimestamp(file_stats.st_ctime).astimezone()
|
||||
modified = datetime.fromtimestamp(file_stats.st_mtime).astimezone()
|
||||
file_metadata = await self.file_service.get_file_metadata(path)
|
||||
created = file_metadata.created_at
|
||||
modified = file_metadata.modified_at
|
||||
|
||||
# entity markdown will always contain front matter, so it can be used up create/update the entity
|
||||
entity_markdown = await self.entity_parser.parse_file(path)
|
||||
# Parse markdown content with file metadata (avoids redundant file read/stat)
|
||||
# This enables cloud implementations (S3FileService) to provide metadata from head_object
|
||||
abs_path = self.file_service.base_path / path
|
||||
entity_markdown = await self.entity_parser.parse_markdown_content(
|
||||
file_path=abs_path,
|
||||
content=file_content,
|
||||
mtime=file_metadata.modified_at.timestamp(),
|
||||
ctime=file_metadata.created_at.timestamp(),
|
||||
)
|
||||
|
||||
# if the file contains frontmatter, resolve a permalink (unless disabled)
|
||||
if file_contains_frontmatter and not self.app_config.disable_permalinks:
|
||||
@@ -723,8 +700,8 @@ class SyncService:
|
||||
"checksum": final_checksum,
|
||||
"created_at": created,
|
||||
"updated_at": modified,
|
||||
"mtime": file_stats.st_mtime,
|
||||
"size": file_stats.st_size,
|
||||
"mtime": file_metadata.modified_at.timestamp(),
|
||||
"size": file_metadata.size,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -737,7 +714,6 @@ class SyncService:
|
||||
# Return the final checksum to ensure everything is consistent
|
||||
return entity, final_checksum
|
||||
|
||||
@logfire.instrument()
|
||||
async def sync_regular_file(self, path: str, new: bool = True) -> Tuple[Optional[Entity], str]:
|
||||
"""Sync a non-markdown file with basic tracking.
|
||||
|
||||
@@ -754,9 +730,9 @@ class SyncService:
|
||||
await self.entity_service.resolve_permalink(path, skip_conflict_check=True)
|
||||
|
||||
# get file timestamps
|
||||
file_stats = self.file_service.file_stats(path)
|
||||
created = datetime.fromtimestamp(file_stats.st_ctime).astimezone()
|
||||
modified = datetime.fromtimestamp(file_stats.st_mtime).astimezone()
|
||||
file_metadata = await self.file_service.get_file_metadata(path)
|
||||
created = file_metadata.created_at
|
||||
modified = file_metadata.modified_at
|
||||
|
||||
# get mime type
|
||||
content_type = self.file_service.content_type(path)
|
||||
@@ -772,8 +748,8 @@ class SyncService:
|
||||
created_at=created,
|
||||
updated_at=modified,
|
||||
content_type=content_type,
|
||||
mtime=file_stats.st_mtime,
|
||||
size=file_stats.st_size,
|
||||
mtime=file_metadata.modified_at.timestamp(),
|
||||
size=file_metadata.size,
|
||||
)
|
||||
)
|
||||
return entity, checksum
|
||||
@@ -789,15 +765,15 @@ class SyncService:
|
||||
logger.error(f"Entity not found after constraint violation, path={path}")
|
||||
raise ValueError(f"Entity not found after constraint violation: {path}")
|
||||
|
||||
# Re-get file stats since we're in update path
|
||||
file_stats_for_update = self.file_service.file_stats(path)
|
||||
# Re-get file metadata since we're in update path
|
||||
file_metadata_for_update = await self.file_service.get_file_metadata(path)
|
||||
updated = await self.entity_repository.update(
|
||||
entity.id,
|
||||
{
|
||||
"file_path": path,
|
||||
"checksum": checksum,
|
||||
"mtime": file_stats_for_update.st_mtime,
|
||||
"size": file_stats_for_update.st_size,
|
||||
"mtime": file_metadata_for_update.modified_at.timestamp(),
|
||||
"size": file_metadata_for_update.size,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -811,8 +787,8 @@ class SyncService:
|
||||
raise
|
||||
else:
|
||||
# Get file timestamps for updating modification time
|
||||
file_stats = self.file_service.file_stats(path)
|
||||
modified = datetime.fromtimestamp(file_stats.st_mtime).astimezone()
|
||||
file_metadata = await self.file_service.get_file_metadata(path)
|
||||
modified = file_metadata.modified_at
|
||||
|
||||
entity = await self.entity_repository.get_by_file_path(path)
|
||||
if entity is None: # pragma: no cover
|
||||
@@ -827,8 +803,8 @@ class SyncService:
|
||||
"file_path": path,
|
||||
"checksum": checksum,
|
||||
"updated_at": modified,
|
||||
"mtime": file_stats.st_mtime,
|
||||
"size": file_stats.st_size,
|
||||
"mtime": file_metadata.modified_at.timestamp(),
|
||||
"size": file_metadata.size,
|
||||
},
|
||||
)
|
||||
|
||||
@@ -838,7 +814,6 @@ class SyncService:
|
||||
|
||||
return updated, checksum
|
||||
|
||||
@logfire.instrument()
|
||||
async def handle_delete(self, file_path: str):
|
||||
"""Handle complete entity deletion including search index cleanup."""
|
||||
|
||||
@@ -870,7 +845,6 @@ class SyncService:
|
||||
else:
|
||||
await self.search_service.delete_by_entity_id(entity.id)
|
||||
|
||||
@logfire.instrument()
|
||||
async def handle_move(self, old_path, new_path):
|
||||
logger.debug("Moving entity", old_path=old_path, new_path=new_path)
|
||||
|
||||
@@ -975,7 +949,6 @@ class SyncService:
|
||||
# update search index
|
||||
await self.search_service.index_entity(updated)
|
||||
|
||||
@logfire.instrument()
|
||||
async def resolve_relations(self, entity_id: int | None = None):
|
||||
"""Try to resolve unresolved relations.
|
||||
|
||||
@@ -1026,16 +999,27 @@ class SyncService:
|
||||
"to_name": resolved_entity.title,
|
||||
},
|
||||
)
|
||||
except IntegrityError: # pragma: no cover
|
||||
# update search index only on successful resolution
|
||||
await self.search_service.index_entity(resolved_entity)
|
||||
except IntegrityError:
|
||||
# IntegrityError means a relation with this (from_id, to_id, relation_type)
|
||||
# already exists. The UPDATE was rolled back, so our unresolved relation
|
||||
# (to_id=NULL) still exists in the database. We delete it because:
|
||||
# 1. It's redundant - a resolved relation already captures this relationship
|
||||
# 2. If we don't delete it, future syncs will try to resolve it again
|
||||
# and get the same IntegrityError
|
||||
logger.debug(
|
||||
"Ignoring duplicate relation "
|
||||
"Deleting duplicate unresolved relation "
|
||||
f"relation_id={relation.id} "
|
||||
f"from_id={relation.from_id} "
|
||||
f"to_name={relation.to_name}"
|
||||
f"to_name={relation.to_name} "
|
||||
f"resolved_to_id={resolved_entity.id}"
|
||||
)
|
||||
|
||||
# update search index
|
||||
await self.search_service.index_entity(resolved_entity)
|
||||
try:
|
||||
await self.relation_repository.delete(relation.id)
|
||||
except Exception as e:
|
||||
# Log but don't fail - the relation may have been deleted already
|
||||
logger.debug(f"Could not delete duplicate relation {relation.id}: {e}")
|
||||
|
||||
async def _quick_count_files(self, directory: Path) -> int:
|
||||
"""Fast file count using find command.
|
||||
@@ -1063,8 +1047,6 @@ class SyncService:
|
||||
f"error: {error_msg}. Falling back to manual count. "
|
||||
f"This will slow down watermark detection!"
|
||||
)
|
||||
# Track optimization failures for visibility
|
||||
logfire.metric_counter("sync.scan.file_count_failure").add(1)
|
||||
# Fallback: count using scan_directory
|
||||
count = 0
|
||||
async for _ in self.scan_directory(directory):
|
||||
@@ -1105,8 +1087,6 @@ class SyncService:
|
||||
f"error: {error_msg}. Falling back to full scan. "
|
||||
f"This will cause slow syncs on large projects!"
|
||||
)
|
||||
# Track optimization failures for visibility
|
||||
logfire.metric_counter("sync.scan.optimization_failure").add(1)
|
||||
# Fallback to full scan
|
||||
return await self._scan_directory_full(directory)
|
||||
|
||||
@@ -1213,7 +1193,7 @@ async def get_sync_service(project: Project) -> SyncService: # pragma: no cover
|
||||
entity_repository = EntityRepository(session_maker, project_id=project.id)
|
||||
observation_repository = ObservationRepository(session_maker, project_id=project.id)
|
||||
relation_repository = RelationRepository(session_maker, project_id=project.id)
|
||||
search_repository = SearchRepository(session_maker, project_id=project.id)
|
||||
search_repository = create_search_repository(session_maker, project_id=project.id)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
|
||||
# Initialize services
|
||||
|
||||
@@ -5,7 +5,10 @@ import os
|
||||
from collections import defaultdict
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Set, Sequence
|
||||
from typing import List, Optional, Set, Sequence, Callable, Awaitable, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from basic_memory.sync.sync_service import SyncService
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, WATCH_STATUS_JSON
|
||||
from basic_memory.ignore_utils import load_gitignore_patterns, should_ignore_path
|
||||
@@ -71,12 +74,17 @@ class WatchServiceState(BaseModel):
|
||||
self.last_error = datetime.now()
|
||||
|
||||
|
||||
# Type alias for sync service factory function
|
||||
SyncServiceFactory = Callable[[Project], Awaitable["SyncService"]]
|
||||
|
||||
|
||||
class WatchService:
|
||||
def __init__(
|
||||
self,
|
||||
app_config: BasicMemoryConfig,
|
||||
project_repository: ProjectRepository,
|
||||
quiet: bool = False,
|
||||
sync_service_factory: Optional[SyncServiceFactory] = None,
|
||||
):
|
||||
self.app_config = app_config
|
||||
self.project_repository = project_repository
|
||||
@@ -84,10 +92,20 @@ class WatchService:
|
||||
self.status_path = Path.home() / ".basic-memory" / WATCH_STATUS_JSON
|
||||
self.status_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
self._ignore_patterns_cache: dict[Path, Set[str]] = {}
|
||||
self._sync_service_factory = sync_service_factory
|
||||
|
||||
# quiet mode for mcp so it doesn't mess up stdout
|
||||
self.console = Console(quiet=quiet)
|
||||
|
||||
async def _get_sync_service(self, project: Project) -> "SyncService":
|
||||
"""Get sync service for a project, using factory if provided."""
|
||||
if self._sync_service_factory:
|
||||
return await self._sync_service_factory(project)
|
||||
# Fall back to default factory
|
||||
from basic_memory.sync.sync_service import get_sync_service
|
||||
|
||||
return await get_sync_service(project)
|
||||
|
||||
async def _schedule_restart(self, stop_event: asyncio.Event):
|
||||
"""Schedule a restart of the watch service after the configured interval."""
|
||||
await asyncio.sleep(self.app_config.watch_project_reload_interval)
|
||||
@@ -233,9 +251,6 @@ class WatchService:
|
||||
|
||||
async def handle_changes(self, project: Project, changes: Set[FileChange]) -> None:
|
||||
"""Process a batch of file changes"""
|
||||
# avoid circular imports
|
||||
from basic_memory.sync.sync_service import get_sync_service
|
||||
|
||||
# Check if project still exists in configuration before processing
|
||||
# This prevents deleted projects from being recreated by background sync
|
||||
from basic_memory.config import ConfigManager
|
||||
@@ -250,7 +265,7 @@ class WatchService:
|
||||
)
|
||||
return
|
||||
|
||||
sync_service = await get_sync_service(project)
|
||||
sync_service = await self._get_sync_service(project)
|
||||
file_service = sync_service.file_service
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
+89
-70
@@ -5,9 +5,9 @@ import os
|
||||
import logging
|
||||
import re
|
||||
import sys
|
||||
from datetime import datetime
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Optional, Protocol, Union, runtime_checkable, List
|
||||
from typing import Protocol, Union, runtime_checkable, List
|
||||
|
||||
from loguru import logger
|
||||
from unidecode import unidecode
|
||||
@@ -67,19 +67,20 @@ class PathLike(Protocol):
|
||||
# This preserves compatibility with existing code while we migrate
|
||||
FilePath = Union[Path, str]
|
||||
|
||||
# Disable the "Queue is full" warning
|
||||
logging.getLogger("opentelemetry.sdk.metrics._internal.instrument").setLevel(logging.ERROR)
|
||||
|
||||
|
||||
def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: bool = True) -> str:
|
||||
"""Generate a stable permalink from a file path.
|
||||
|
||||
Args:
|
||||
file_path: Original file path (str, Path, or PathLike)
|
||||
split_extension: Whether to split off and discard file extensions.
|
||||
When True, uses mimetypes to detect real extensions.
|
||||
When False, preserves all content including periods.
|
||||
|
||||
Returns:
|
||||
Normalized permalink that matches validation rules. Converts spaces and underscores
|
||||
to hyphens for consistency. Preserves non-ASCII characters like Chinese.
|
||||
Preserves periods in version numbers (e.g., "2.0.0") when they're not real file extensions.
|
||||
|
||||
Examples:
|
||||
>>> generate_permalink("docs/My Feature.md")
|
||||
@@ -90,12 +91,26 @@ def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: b
|
||||
'design/unified-model-refactor'
|
||||
>>> generate_permalink("中文/测试文档.md")
|
||||
'中文/测试文档'
|
||||
>>> generate_permalink("Version 2.0.0")
|
||||
'version-2.0.0'
|
||||
"""
|
||||
# Convert Path to string if needed
|
||||
path_str = Path(str(file_path)).as_posix()
|
||||
|
||||
# Remove extension (for now, possibly)
|
||||
(base, extension) = os.path.splitext(path_str)
|
||||
# Only split extension if there's a real file extension
|
||||
# Use mimetypes to detect real extensions, avoiding misinterpreting periods in version numbers
|
||||
import mimetypes
|
||||
|
||||
mime_type, _ = mimetypes.guess_type(path_str)
|
||||
has_real_extension = mime_type is not None
|
||||
|
||||
if has_real_extension and split_extension:
|
||||
# Real file extension detected - split it off
|
||||
(base, extension) = os.path.splitext(path_str)
|
||||
else:
|
||||
# No real extension or split_extension=False - process the whole string
|
||||
base = path_str
|
||||
extension = ""
|
||||
|
||||
# Check if we have CJK characters that should be preserved
|
||||
# CJK ranges: \u4e00-\u9fff (CJK Unified Ideographs), \u3000-\u303f (CJK symbols),
|
||||
@@ -147,9 +162,9 @@ def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: b
|
||||
# Remove apostrophes entirely (don't replace with hyphens)
|
||||
text_no_apostrophes = text_with_hyphens.replace("'", "")
|
||||
|
||||
# Replace unsafe chars with hyphens, but preserve CJK characters
|
||||
# Replace unsafe chars with hyphens, but preserve CJK characters and periods
|
||||
clean_text = re.sub(
|
||||
r"[^a-z0-9\u4e00-\u9fff\u3000-\u303f\u3400-\u4dbf/\-]", "-", text_no_apostrophes
|
||||
r"[^a-z0-9\u4e00-\u9fff\u3000-\u303f\u3400-\u4dbf/\-\.]", "-", text_no_apostrophes
|
||||
)
|
||||
else:
|
||||
# Original ASCII-only processing for backward compatibility
|
||||
@@ -168,8 +183,8 @@ def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: b
|
||||
# Remove apostrophes entirely (don't replace with hyphens)
|
||||
text_no_apostrophes = text_with_hyphens.replace("'", "")
|
||||
|
||||
# Replace remaining invalid chars with hyphens
|
||||
clean_text = re.sub(r"[^a-z0-9/\-]", "-", text_no_apostrophes)
|
||||
# Replace remaining invalid chars with hyphens, preserving periods
|
||||
clean_text = re.sub(r"[^a-z0-9/\-\.]", "-", text_no_apostrophes)
|
||||
|
||||
# Collapse multiple hyphens
|
||||
clean_text = re.sub(r"-+", "-", clean_text)
|
||||
@@ -188,29 +203,35 @@ def generate_permalink(file_path: Union[Path, str, PathLike], split_extension: b
|
||||
|
||||
|
||||
def setup_logging(
|
||||
env: str,
|
||||
home_dir: Path,
|
||||
log_file: Optional[str] = None,
|
||||
log_level: str = "INFO",
|
||||
console: bool = True,
|
||||
log_to_file: bool = False,
|
||||
log_to_stdout: bool = False,
|
||||
structured_context: bool = False,
|
||||
) -> None: # pragma: no cover
|
||||
"""
|
||||
Configure logging for the application.
|
||||
"""Configure logging with explicit settings.
|
||||
|
||||
This function provides a simple, explicit interface for configuring logging.
|
||||
Each entry point (CLI, MCP, API) should call this with appropriate settings.
|
||||
|
||||
Args:
|
||||
env: The environment name (dev, test, prod)
|
||||
home_dir: The root directory for the application
|
||||
log_file: The name of the log file to write to
|
||||
log_level: The logging level to use
|
||||
console: Whether to log to the console
|
||||
log_level: DEBUG, INFO, WARNING, ERROR
|
||||
log_to_file: Write to ~/.basic-memory/basic-memory.log with rotation
|
||||
log_to_stdout: Write to stderr (for Docker/cloud deployments)
|
||||
structured_context: Bind tenant_id, fly_region, etc. for cloud observability
|
||||
"""
|
||||
# Remove default handler and any existing handlers
|
||||
logger.remove()
|
||||
|
||||
# Add file handler if we are not running tests and a log file is specified
|
||||
if log_file and env != "test":
|
||||
# Setup file logger
|
||||
log_path = home_dir / log_file
|
||||
# In test mode, only log to stdout regardless of settings
|
||||
env = os.getenv("BASIC_MEMORY_ENV", "dev")
|
||||
if env == "test":
|
||||
logger.add(sys.stderr, level=log_level, backtrace=True, diagnose=True, colorize=True)
|
||||
return
|
||||
|
||||
# Add file handler with rotation
|
||||
if log_to_file:
|
||||
log_path = Path.home() / ".basic-memory" / "basic-memory.log"
|
||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
logger.add(
|
||||
str(log_path),
|
||||
level=log_level,
|
||||
@@ -218,42 +239,28 @@ def setup_logging(
|
||||
retention="10 days",
|
||||
backtrace=True,
|
||||
diagnose=True,
|
||||
enqueue=True,
|
||||
enqueue=True, # Thread-safe async logging
|
||||
colorize=False,
|
||||
)
|
||||
|
||||
# Add console logger if requested or in test mode
|
||||
if env == "test" or console:
|
||||
# Add stdout handler (for Docker/cloud)
|
||||
if log_to_stdout:
|
||||
logger.add(sys.stderr, level=log_level, backtrace=True, diagnose=True, colorize=True)
|
||||
|
||||
logger.info(f"ENV: '{env}' Log level: '{log_level}' Logging to {log_file}")
|
||||
|
||||
# Bind environment context for structured logging (works in both local and cloud)
|
||||
tenant_id = os.getenv("BASIC_MEMORY_TENANT_ID", "local")
|
||||
fly_app_name = os.getenv("FLY_APP_NAME", "local")
|
||||
fly_machine_id = os.getenv("FLY_MACHINE_ID", "local")
|
||||
fly_region = os.getenv("FLY_REGION", "local")
|
||||
|
||||
logger.configure(
|
||||
extra={
|
||||
"tenant_id": tenant_id,
|
||||
"fly_app_name": fly_app_name,
|
||||
"fly_machine_id": fly_machine_id,
|
||||
"fly_region": fly_region,
|
||||
}
|
||||
)
|
||||
# Bind structured context for cloud observability
|
||||
if structured_context:
|
||||
logger.configure(
|
||||
extra={
|
||||
"tenant_id": os.getenv("BASIC_MEMORY_TENANT_ID", "local"),
|
||||
"fly_app_name": os.getenv("FLY_APP_NAME", "local"),
|
||||
"fly_machine_id": os.getenv("FLY_MACHINE_ID", "local"),
|
||||
"fly_region": os.getenv("FLY_REGION", "local"),
|
||||
}
|
||||
)
|
||||
|
||||
# Reduce noise from third-party libraries
|
||||
noisy_loggers = {
|
||||
# HTTP client logs
|
||||
"httpx": logging.WARNING,
|
||||
# File watching logs
|
||||
"watchfiles.main": logging.WARNING,
|
||||
}
|
||||
|
||||
# Set log levels for noisy loggers
|
||||
for logger_name, level in noisy_loggers.items():
|
||||
logging.getLogger(logger_name).setLevel(level)
|
||||
logging.getLogger("httpx").setLevel(logging.WARNING)
|
||||
logging.getLogger("watchfiles.main").setLevel(logging.WARNING)
|
||||
|
||||
|
||||
def parse_tags(tags: Union[List[str], str, None]) -> List[str]:
|
||||
@@ -322,7 +329,7 @@ def normalize_file_path_for_comparison(file_path: str) -> str:
|
||||
This function normalizes file paths to help detect potential conflicts:
|
||||
- Converts to lowercase for case-insensitive comparison
|
||||
- Normalizes Unicode characters
|
||||
- Handles path separators consistently
|
||||
- Converts backslashes to forward slashes for cross-platform consistency
|
||||
|
||||
Args:
|
||||
file_path: The file path to normalize
|
||||
@@ -331,19 +338,15 @@ def normalize_file_path_for_comparison(file_path: str) -> str:
|
||||
Normalized file path for comparison purposes
|
||||
"""
|
||||
import unicodedata
|
||||
from pathlib import PureWindowsPath
|
||||
|
||||
# Convert to lowercase for case-insensitive comparison
|
||||
normalized = file_path.lower()
|
||||
# Use PureWindowsPath to ensure backslashes are treated as separators
|
||||
# regardless of current platform, then convert to POSIX-style
|
||||
normalized = PureWindowsPath(file_path).as_posix().lower()
|
||||
|
||||
# Normalize Unicode characters (NFD normalization)
|
||||
normalized = unicodedata.normalize("NFD", normalized)
|
||||
|
||||
# Replace path separators with forward slashes
|
||||
normalized = normalized.replace("\\", "/")
|
||||
|
||||
# Remove multiple slashes
|
||||
normalized = re.sub(r"/+", "/", normalized)
|
||||
|
||||
return normalized
|
||||
|
||||
|
||||
@@ -427,21 +430,37 @@ def validate_project_path(path: str, project_path: Path) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def ensure_timezone_aware(dt: datetime) -> datetime:
|
||||
"""Ensure a datetime is timezone-aware using system timezone.
|
||||
def ensure_timezone_aware(dt: datetime, cloud_mode: bool | None = None) -> datetime:
|
||||
"""Ensure a datetime is timezone-aware.
|
||||
|
||||
If the datetime is naive, convert it to timezone-aware using the system's local timezone.
|
||||
If it's already timezone-aware, return it unchanged.
|
||||
If the datetime is naive, convert it to timezone-aware. The interpretation
|
||||
depends on cloud_mode:
|
||||
- In cloud mode (PostgreSQL/asyncpg): naive datetimes are interpreted as UTC
|
||||
- In local mode (SQLite): naive datetimes are interpreted as local time
|
||||
|
||||
asyncpg uses binary protocol which returns timestamps in UTC but as naive
|
||||
datetimes. In cloud deployments, cloud_mode=True handles this correctly.
|
||||
|
||||
Args:
|
||||
dt: The datetime to ensure is timezone-aware
|
||||
cloud_mode: Optional explicit cloud_mode setting. If None, loads from config.
|
||||
|
||||
Returns:
|
||||
A timezone-aware datetime
|
||||
"""
|
||||
if dt.tzinfo is None:
|
||||
# Naive datetime - assume it's in local time and add timezone
|
||||
return dt.astimezone()
|
||||
# Determine cloud_mode: use explicit parameter if provided, otherwise load from config
|
||||
if cloud_mode is None:
|
||||
from basic_memory.config import ConfigManager
|
||||
|
||||
cloud_mode = ConfigManager().config.cloud_mode_enabled
|
||||
|
||||
if cloud_mode:
|
||||
# Cloud/PostgreSQL mode: naive datetimes from asyncpg are already UTC
|
||||
return dt.replace(tzinfo=timezone.utc)
|
||||
else:
|
||||
# Local/SQLite mode: naive datetimes are in local time
|
||||
return dt.astimezone()
|
||||
else:
|
||||
# Already timezone-aware
|
||||
return dt
|
||||
|
||||
@@ -5,13 +5,13 @@ from pathlib import Path
|
||||
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from basic_memory.cli.main import app
|
||||
from basic_memory.cli.main import app as cli_app
|
||||
|
||||
|
||||
def test_project_list(app_config, test_project, config_manager):
|
||||
def test_project_list(app, app_config, test_project, config_manager):
|
||||
"""Test 'bm project list' command shows projects."""
|
||||
runner = CliRunner()
|
||||
result = runner.invoke(app, ["project", "list"])
|
||||
result = runner.invoke(cli_app, ["project", "list"])
|
||||
|
||||
if result.exit_code != 0:
|
||||
print(f"STDOUT: {result.stdout}")
|
||||
@@ -22,10 +22,10 @@ def test_project_list(app_config, test_project, config_manager):
|
||||
assert "[X]" in result.stdout # default marker
|
||||
|
||||
|
||||
def test_project_info(app_config, test_project, config_manager):
|
||||
def test_project_info(app, app_config, test_project, config_manager):
|
||||
"""Test 'bm project info' command shows project details."""
|
||||
runner = CliRunner()
|
||||
result = runner.invoke(app, ["project", "info", "test-project"])
|
||||
result = runner.invoke(cli_app, ["project", "info", "test-project"])
|
||||
|
||||
if result.exit_code != 0:
|
||||
print(f"STDOUT: {result.stdout}")
|
||||
@@ -36,12 +36,12 @@ def test_project_info(app_config, test_project, config_manager):
|
||||
assert "Statistics" in result.stdout
|
||||
|
||||
|
||||
def test_project_info_json(app_config, test_project, config_manager):
|
||||
def test_project_info_json(app, app_config, test_project, config_manager):
|
||||
"""Test 'bm project info --json' command outputs valid JSON."""
|
||||
import json
|
||||
|
||||
runner = CliRunner()
|
||||
result = runner.invoke(app, ["project", "info", "test-project", "--json"])
|
||||
result = runner.invoke(cli_app, ["project", "info", "test-project", "--json"])
|
||||
|
||||
if result.exit_code != 0:
|
||||
print(f"STDOUT: {result.stdout}")
|
||||
@@ -55,7 +55,7 @@ def test_project_info_json(app_config, test_project, config_manager):
|
||||
assert "system" in data
|
||||
|
||||
|
||||
def test_project_add_and_remove(app_config, config_manager):
|
||||
def test_project_add_and_remove(app, app_config, config_manager):
|
||||
"""Test adding and removing a project."""
|
||||
runner = CliRunner()
|
||||
|
||||
@@ -65,7 +65,7 @@ def test_project_add_and_remove(app_config, config_manager):
|
||||
new_project_path.mkdir()
|
||||
|
||||
# Add project
|
||||
result = runner.invoke(app, ["project", "add", "new-project", str(new_project_path)])
|
||||
result = runner.invoke(cli_app, ["project", "add", "new-project", str(new_project_path)])
|
||||
|
||||
if result.exit_code != 0:
|
||||
print(f"STDOUT: {result.stdout}")
|
||||
@@ -77,17 +77,17 @@ def test_project_add_and_remove(app_config, config_manager):
|
||||
)
|
||||
|
||||
# Verify it shows up in list
|
||||
result = runner.invoke(app, ["project", "list"])
|
||||
result = runner.invoke(cli_app, ["project", "list"])
|
||||
assert result.exit_code == 0
|
||||
assert "new-project" in result.stdout
|
||||
|
||||
# Remove project
|
||||
result = runner.invoke(app, ["project", "remove", "new-project"])
|
||||
result = runner.invoke(cli_app, ["project", "remove", "new-project"])
|
||||
assert result.exit_code == 0
|
||||
assert "removed" in result.stdout.lower() or "deleted" in result.stdout.lower()
|
||||
|
||||
|
||||
def test_project_set_default(app_config, config_manager):
|
||||
def test_project_set_default(app, app_config, config_manager):
|
||||
"""Test setting default project."""
|
||||
runner = CliRunner()
|
||||
|
||||
@@ -97,14 +97,16 @@ def test_project_set_default(app_config, config_manager):
|
||||
new_project_path.mkdir()
|
||||
|
||||
# Add a second project
|
||||
result = runner.invoke(app, ["project", "add", "another-project", str(new_project_path)])
|
||||
result = runner.invoke(
|
||||
cli_app, ["project", "add", "another-project", str(new_project_path)]
|
||||
)
|
||||
if result.exit_code != 0:
|
||||
print(f"STDOUT: {result.stdout}")
|
||||
print(f"STDERR: {result.stderr}")
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Set as default
|
||||
result = runner.invoke(app, ["project", "default", "another-project"])
|
||||
result = runner.invoke(cli_app, ["project", "default", "another-project"])
|
||||
if result.exit_code != 0:
|
||||
print(f"STDOUT: {result.stdout}")
|
||||
print(f"STDERR: {result.stderr}")
|
||||
@@ -112,10 +114,52 @@ def test_project_set_default(app_config, config_manager):
|
||||
assert "default" in result.stdout.lower()
|
||||
|
||||
# Verify in list
|
||||
result = runner.invoke(app, ["project", "list"])
|
||||
result = runner.invoke(cli_app, ["project", "list"])
|
||||
assert result.exit_code == 0
|
||||
# The new project should have the [X] marker now
|
||||
lines = result.stdout.split("\n")
|
||||
for line in lines:
|
||||
if "another-project" in line:
|
||||
assert "[X]" in line
|
||||
|
||||
|
||||
def test_remove_main_project(app, app_config, config_manager):
|
||||
"""Test that removing main project then listing projects prevents main from reappearing (issue #397)."""
|
||||
runner = CliRunner()
|
||||
|
||||
# Create separate temp dirs for each project
|
||||
with (
|
||||
tempfile.TemporaryDirectory() as main_dir,
|
||||
tempfile.TemporaryDirectory() as new_default_dir,
|
||||
):
|
||||
main_path = Path(main_dir)
|
||||
new_default_path = Path(new_default_dir)
|
||||
|
||||
# Ensure main exists
|
||||
result = runner.invoke(cli_app, ["project", "list"])
|
||||
if "main" not in result.stdout:
|
||||
result = runner.invoke(cli_app, ["project", "add", "main", str(main_path)])
|
||||
print(result.stdout)
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Confirm main is present
|
||||
result = runner.invoke(cli_app, ["project", "list"])
|
||||
assert "main" in result.stdout
|
||||
|
||||
# Add a second project
|
||||
result = runner.invoke(cli_app, ["project", "add", "new_default", str(new_default_path)])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Set new_default as default (if needed)
|
||||
result = runner.invoke(cli_app, ["project", "default", "new_default"])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Remove main
|
||||
result = runner.invoke(cli_app, ["project", "remove", "main"])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Confirm only new_default exists and main does not
|
||||
result = runner.invoke(cli_app, ["project", "list"])
|
||||
assert result.exit_code == 0
|
||||
assert "main" not in result.stdout
|
||||
assert "new_default" in result.stdout
|
||||
|
||||
+146
-26
@@ -50,17 +50,23 @@ The `app` fixture ensures FastAPI dependency overrides are active, and
|
||||
`mcp_server` provides the MCP server with proper project session initialization.
|
||||
"""
|
||||
|
||||
from typing import AsyncGenerator
|
||||
import os
|
||||
from typing import AsyncGenerator, Literal
|
||||
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
from pathlib import Path
|
||||
from sqlalchemy import text
|
||||
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||
from sqlalchemy.pool import NullPool
|
||||
from testcontainers.postgres import PostgresContainer
|
||||
|
||||
from httpx import AsyncClient, ASGITransport
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, ProjectConfig, ConfigManager
|
||||
from basic_memory.config import BasicMemoryConfig, ProjectConfig, ConfigManager, DatabaseBackend
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.models.base import Base
|
||||
from basic_memory.repository.project_repository import ProjectRepository
|
||||
from fastapi import FastAPI
|
||||
|
||||
@@ -71,24 +77,109 @@ from basic_memory.deps import get_project_config, get_engine_factory, get_app_co
|
||||
from basic_memory.mcp import tools # noqa: F401
|
||||
|
||||
|
||||
@pytest_asyncio.fixture(scope="function")
|
||||
async def engine_factory(tmp_path):
|
||||
"""Create a SQLite file engine factory for integration testing."""
|
||||
db_path = tmp_path / "test.db"
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (
|
||||
engine,
|
||||
session_maker,
|
||||
):
|
||||
# Initialize database schema
|
||||
from basic_memory.models.base import Base
|
||||
# =============================================================================
|
||||
# Database Backend Selection (env var approach)
|
||||
# =============================================================================
|
||||
# By default, integration tests run against SQLite.
|
||||
# Set BASIC_MEMORY_TEST_POSTGRES=1 to run against Postgres (uses testcontainers).
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def db_backend() -> Literal["sqlite", "postgres"]:
|
||||
"""Determine database backend from environment variable.
|
||||
|
||||
Default: sqlite
|
||||
Set BASIC_MEMORY_TEST_POSTGRES=1 to use postgres
|
||||
"""
|
||||
if os.environ.get("BASIC_MEMORY_TEST_POSTGRES", "").lower() in ("1", "true", "yes"):
|
||||
return "postgres"
|
||||
return "sqlite"
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def postgres_container(db_backend):
|
||||
"""Session-scoped Postgres container for integration tests.
|
||||
|
||||
Uses testcontainers to spin up a real Postgres instance.
|
||||
Only starts if db_backend is "postgres".
|
||||
"""
|
||||
if db_backend != "postgres":
|
||||
yield None
|
||||
return
|
||||
|
||||
with PostgresContainer("postgres:16-alpine") as postgres:
|
||||
yield postgres
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def engine_factory(
|
||||
app_config,
|
||||
config_manager,
|
||||
db_backend: Literal["sqlite", "postgres"],
|
||||
postgres_container,
|
||||
tmp_path,
|
||||
) -> AsyncGenerator[tuple, None]:
|
||||
"""Create engine and session factory for the configured database backend."""
|
||||
from basic_memory.models.search import (
|
||||
CREATE_SEARCH_INDEX,
|
||||
CREATE_POSTGRES_SEARCH_INDEX_TABLE,
|
||||
CREATE_POSTGRES_SEARCH_INDEX_FTS,
|
||||
CREATE_POSTGRES_SEARCH_INDEX_METADATA,
|
||||
)
|
||||
from basic_memory import db
|
||||
|
||||
if db_backend == "postgres":
|
||||
# Postgres mode using testcontainers
|
||||
sync_url = postgres_container.get_connection_url()
|
||||
async_url = sync_url.replace("postgresql+psycopg2", "postgresql+asyncpg")
|
||||
|
||||
engine = create_async_engine(
|
||||
async_url,
|
||||
echo=False,
|
||||
poolclass=NullPool,
|
||||
)
|
||||
|
||||
session_maker = async_sessionmaker(
|
||||
bind=engine,
|
||||
class_=AsyncSession,
|
||||
expire_on_commit=False,
|
||||
autoflush=False,
|
||||
)
|
||||
|
||||
# Drop and recreate all tables for test isolation
|
||||
async with engine.begin() as conn:
|
||||
await conn.execute(text("DROP TABLE IF EXISTS search_index CASCADE"))
|
||||
await conn.run_sync(Base.metadata.drop_all)
|
||||
await conn.run_sync(Base.metadata.create_all)
|
||||
# asyncpg requires separate execute calls for each statement
|
||||
await conn.execute(CREATE_POSTGRES_SEARCH_INDEX_TABLE)
|
||||
await conn.execute(CREATE_POSTGRES_SEARCH_INDEX_FTS)
|
||||
await conn.execute(CREATE_POSTGRES_SEARCH_INDEX_METADATA)
|
||||
|
||||
yield engine, session_maker
|
||||
|
||||
await engine.dispose()
|
||||
|
||||
@pytest_asyncio.fixture(scope="function")
|
||||
else:
|
||||
# SQLite: Create fresh database (fast with tmp files)
|
||||
db_path = tmp_path / "test.db"
|
||||
db_type = DatabaseType.FILESYSTEM
|
||||
|
||||
async with engine_session_factory(db_path, db_type) as (engine, session_maker):
|
||||
# Create all tables via ORM
|
||||
async with engine.begin() as conn:
|
||||
await conn.run_sync(Base.metadata.create_all)
|
||||
|
||||
# Drop any SearchIndex ORM table, then create FTS5 virtual table
|
||||
async with db.scoped_session(session_maker) as session:
|
||||
await session.execute(text("DROP TABLE IF EXISTS search_index"))
|
||||
await session.execute(CREATE_SEARCH_INDEX)
|
||||
await session.commit()
|
||||
|
||||
yield engine, session_maker
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def test_project(config_home, engine_factory) -> Project:
|
||||
"""Create a test project."""
|
||||
project_data = {
|
||||
@@ -113,14 +204,31 @@ def config_home(tmp_path, monkeypatch) -> Path:
|
||||
return tmp_path
|
||||
|
||||
|
||||
@pytest.fixture(scope="function", autouse=True)
|
||||
def app_config(config_home, tmp_path, monkeypatch) -> BasicMemoryConfig:
|
||||
@pytest.fixture
|
||||
def app_config(
|
||||
config_home,
|
||||
db_backend: Literal["sqlite", "postgres"],
|
||||
postgres_container,
|
||||
tmp_path,
|
||||
monkeypatch,
|
||||
) -> BasicMemoryConfig:
|
||||
"""Create test app configuration."""
|
||||
# Disable cloud mode for CLI tests
|
||||
monkeypatch.setenv("BASIC_MEMORY_CLOUD_MODE", "false")
|
||||
|
||||
# Create a basic config with test-project like unit tests do
|
||||
projects = {"test-project": str(config_home)}
|
||||
|
||||
# Configure database backend based on env var
|
||||
if db_backend == "postgres":
|
||||
database_backend = DatabaseBackend.POSTGRES
|
||||
# Get URL from testcontainer and convert to asyncpg driver
|
||||
sync_url = postgres_container.get_connection_url()
|
||||
database_url = sync_url.replace("postgresql+psycopg2", "postgresql+asyncpg")
|
||||
else:
|
||||
database_backend = DatabaseBackend.SQLITE
|
||||
database_url = None
|
||||
|
||||
app_config = BasicMemoryConfig(
|
||||
env="test",
|
||||
projects=projects,
|
||||
@@ -128,12 +236,19 @@ def app_config(config_home, tmp_path, monkeypatch) -> BasicMemoryConfig:
|
||||
default_project_mode=False, # Match real-world usage - tools must pass explicit project
|
||||
update_permalinks_on_move=True,
|
||||
cloud_mode=False, # Explicitly disable cloud mode
|
||||
database_backend=database_backend,
|
||||
database_url=database_url,
|
||||
)
|
||||
return app_config
|
||||
|
||||
|
||||
@pytest.fixture(scope="function", autouse=True)
|
||||
@pytest.fixture
|
||||
def config_manager(app_config: BasicMemoryConfig, config_home) -> ConfigManager:
|
||||
# Invalidate config cache to ensure clean state for each test
|
||||
from basic_memory import config as config_module
|
||||
|
||||
config_module._CONFIG_CACHE = None
|
||||
|
||||
config_manager = ConfigManager()
|
||||
# Update its paths to use the test directory
|
||||
config_manager.config_dir = config_home / ".basic-memory"
|
||||
@@ -145,7 +260,7 @@ def config_manager(app_config: BasicMemoryConfig, config_home) -> ConfigManager:
|
||||
return config_manager
|
||||
|
||||
|
||||
@pytest.fixture(scope="function", autouse=True)
|
||||
@pytest.fixture
|
||||
def project_config(test_project):
|
||||
"""Create test project configuration."""
|
||||
|
||||
@@ -157,7 +272,7 @@ def project_config(test_project):
|
||||
return project_config
|
||||
|
||||
|
||||
@pytest.fixture(scope="function")
|
||||
@pytest.fixture
|
||||
def app(app_config, project_config, engine_factory, test_project, config_manager) -> FastAPI:
|
||||
"""Create test FastAPI application with single project."""
|
||||
|
||||
@@ -172,20 +287,25 @@ def app(app_config, project_config, engine_factory, test_project, config_manager
|
||||
return app
|
||||
|
||||
|
||||
@pytest_asyncio.fixture(scope="function")
|
||||
async def search_service(engine_factory, test_project):
|
||||
"""Create and initialize search service for integration tests."""
|
||||
from basic_memory.repository.search_repository import SearchRepository
|
||||
@pytest_asyncio.fixture
|
||||
async def search_service(engine_factory, test_project, app_config):
|
||||
"""Create and initialize search service for integration tests.
|
||||
|
||||
Uses app_config fixture to determine database backend - no patching needed.
|
||||
"""
|
||||
from basic_memory.repository.entity_repository import EntityRepository
|
||||
from basic_memory.services.file_service import FileService
|
||||
from basic_memory.services.search_service import SearchService
|
||||
from basic_memory.markdown.markdown_processor import MarkdownProcessor
|
||||
from basic_memory.markdown import EntityParser
|
||||
|
||||
from basic_memory.repository.search_repository import create_search_repository
|
||||
|
||||
engine, session_maker = engine_factory
|
||||
|
||||
# Create repositories
|
||||
search_repository = SearchRepository(session_maker, project_id=test_project.id)
|
||||
# Use factory function to create appropriate search repository
|
||||
search_repository = create_search_repository(session_maker, project_id=test_project.id)
|
||||
|
||||
entity_repository = EntityRepository(session_maker, project_id=test_project.id)
|
||||
|
||||
# Create file service
|
||||
@@ -199,7 +319,7 @@ async def search_service(engine_factory, test_project):
|
||||
return service
|
||||
|
||||
|
||||
@pytest.fixture(scope="function")
|
||||
@pytest.fixture
|
||||
def mcp_server(config_manager, search_service):
|
||||
# Import mcp instance
|
||||
from basic_memory.mcp.server import mcp as server
|
||||
@@ -213,7 +333,7 @@ def mcp_server(config_manager, search_service):
|
||||
return server
|
||||
|
||||
|
||||
@pytest_asyncio.fixture(scope="function")
|
||||
@pytest_asyncio.fixture
|
||||
async def client(app: FastAPI) -> AsyncGenerator[AsyncClient, None]:
|
||||
"""Create test client that both MCP and tests will use."""
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://test") as client:
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
"""
|
||||
Integration test for FastAPI lifespan shutdown behavior.
|
||||
|
||||
This test verifies the asyncio cancellation pattern used by the API lifespan:
|
||||
when the background sync task is cancelled during shutdown, it must be *awaited*
|
||||
before database shutdown begins. This prevents "hang on exit" scenarios in
|
||||
`asyncio.run(...)` callers (e.g. CLI/MCP clients using httpx ASGITransport).
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
|
||||
from httpx import ASGITransport, AsyncClient
|
||||
|
||||
|
||||
def test_lifespan_shutdown_awaits_sync_task_cancellation(app, monkeypatch):
|
||||
"""
|
||||
Ensure lifespan shutdown awaits the cancelled background sync task.
|
||||
|
||||
Why this is deterministic:
|
||||
- Cancelling a task does not make it "done" immediately; it becomes done only
|
||||
once the event loop schedules it and it processes the CancelledError.
|
||||
- In the buggy version, shutdown proceeded directly to db.shutdown_db()
|
||||
immediately after calling cancel(), so at *entry* to shutdown_db the task
|
||||
is still not done.
|
||||
- In the fixed version, lifespan does `await sync_task` before shutdown_db,
|
||||
so by the time shutdown_db is called, the task is done (cancelled).
|
||||
"""
|
||||
|
||||
# Import the *module* (not the package-level FastAPI `basic_memory.api.app` export)
|
||||
# so monkeypatching affects the exact symbols referenced inside lifespan().
|
||||
#
|
||||
# Note: `basic_memory/api/__init__.py` re-exports `app`, so `import basic_memory.api.app`
|
||||
# can resolve to the FastAPI instance rather than the `basic_memory.api.app` module.
|
||||
import importlib
|
||||
|
||||
api_app_module = importlib.import_module("basic_memory.api.app")
|
||||
|
||||
# Keep startup cheap: we don't need real DB init for this ordering test.
|
||||
async def _noop_initialize_app(_app_config):
|
||||
return None
|
||||
|
||||
async def _fake_get_or_create_db(*_args, **_kwargs):
|
||||
return object(), object()
|
||||
|
||||
monkeypatch.setattr(api_app_module, "initialize_app", _noop_initialize_app)
|
||||
monkeypatch.setattr(api_app_module.db, "get_or_create_db", _fake_get_or_create_db)
|
||||
|
||||
# Make the sync task long-lived so it must be cancelled on shutdown.
|
||||
async def _fake_initialize_file_sync(_app_config):
|
||||
await asyncio.Event().wait()
|
||||
|
||||
monkeypatch.setattr(api_app_module, "initialize_file_sync", _fake_initialize_file_sync)
|
||||
|
||||
# Assert ordering: shutdown_db must be called only after the sync_task is done.
|
||||
async def _assert_sync_task_done_before_db_shutdown():
|
||||
assert api_app_module.app.state.sync_task is not None
|
||||
assert api_app_module.app.state.sync_task.done()
|
||||
|
||||
monkeypatch.setattr(api_app_module.db, "shutdown_db", _assert_sync_task_done_before_db_shutdown)
|
||||
|
||||
async def _run_client_once():
|
||||
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://test") as client:
|
||||
# Any request is sufficient to trigger lifespan startup/shutdown.
|
||||
await client.get("/__nonexistent__")
|
||||
|
||||
# Use asyncio.run to match the CLI/MCP execution model where loop teardown
|
||||
# would hang if a background task is left running.
|
||||
asyncio.run(_run_client_once())
|
||||
|
||||
|
||||
@@ -77,7 +77,8 @@ async def test_create_project_basic_operation(mcp_server, app, test_project):
|
||||
assert "test-new-project" in create_text
|
||||
assert "Project Details:" in create_text
|
||||
assert "Name: test-new-project" in create_text
|
||||
assert "Path: /tmp/test-new-project" in create_text
|
||||
# Check path contains project name (platform-independent)
|
||||
assert "Path:" in create_text and "test-new-project" in create_text
|
||||
assert "Project is now available for use" in create_text
|
||||
|
||||
# Verify project appears in project list
|
||||
|
||||
@@ -46,3 +46,57 @@ async def test_read_note_after_write(mcp_server, app, test_project):
|
||||
assert "# Test Note" in result_text
|
||||
assert "This is test content." in result_text
|
||||
assert "test/test-note" in result_text # permalink
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_read_note_underscored_folder_by_permalink(mcp_server, app, test_project):
|
||||
"""Test read_note with permalink from underscored folder.
|
||||
|
||||
Reproduces bug #416: read_note fails to find notes when given permalinks
|
||||
from underscored folder names (e.g., _archive/, _drafts/), even though
|
||||
the permalink is copied directly from the note's YAML frontmatter.
|
||||
"""
|
||||
|
||||
async with Client(mcp_server) as client:
|
||||
# Create a note in an underscored folder
|
||||
write_result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": "Example Note",
|
||||
"folder": "_archive/articles",
|
||||
"content": "# Example Note\n\nThis is a test note in an underscored folder.",
|
||||
"tags": "test,archive",
|
||||
},
|
||||
)
|
||||
|
||||
assert len(write_result.content) == 1
|
||||
assert write_result.content[0].type == "text"
|
||||
write_text = write_result.content[0].text
|
||||
|
||||
# Verify the file path includes the underscore
|
||||
assert "_archive/articles/Example Note.md" in write_text
|
||||
|
||||
# Verify the permalink has underscores stripped (this is the expected behavior)
|
||||
assert "archive/articles/example-note" in write_text
|
||||
|
||||
# Now try to read the note using the permalink (without underscores)
|
||||
# This is the exact scenario from the bug report - using the permalink
|
||||
# that was generated in the YAML frontmatter
|
||||
read_result = await client.call_tool(
|
||||
"read_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"identifier": "archive/articles/example-note", # permalink without underscores
|
||||
},
|
||||
)
|
||||
|
||||
# This should succeed - the note should be found by its permalink
|
||||
assert len(read_result.content) == 1
|
||||
assert read_result.content[0].type == "text"
|
||||
result_text = read_result.content[0].text
|
||||
|
||||
# Should contain the note content
|
||||
assert "# Example Note" in result_text
|
||||
assert "This is a test note in an underscored folder." in result_text
|
||||
assert "archive/articles/example-note" in result_text # permalink
|
||||
|
||||
@@ -9,9 +9,10 @@ from textwrap import dedent
|
||||
|
||||
import pytest
|
||||
from fastmcp import Client
|
||||
from unittest.mock import patch
|
||||
|
||||
from basic_memory.config import ConfigManager
|
||||
from basic_memory.schemas.project_info import ProjectItem
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -313,79 +314,68 @@ async def test_write_note_preserve_frontmatter(mcp_server, app, test_project):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_write_note_kebab_filenames_basic(mcp_server, test_project):
|
||||
async def test_write_note_kebab_filenames_basic(mcp_server, app, test_project, app_config):
|
||||
"""Test note creation with kebab_filenames=True and invalid filename characters."""
|
||||
|
||||
config = ConfigManager().config
|
||||
curr_config_val = config.kebab_filenames
|
||||
config.kebab_filenames = True
|
||||
app_config.kebab_filenames = True
|
||||
ConfigManager().save_config(app_config)
|
||||
|
||||
with patch.object(ConfigManager, "config", config):
|
||||
async with Client(mcp_server) as client:
|
||||
result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": "My Note: With/Invalid|Chars?",
|
||||
"folder": "my-folder",
|
||||
"content": "Testing kebab-case and invalid characters.",
|
||||
"tags": "kebab,invalid,filename",
|
||||
},
|
||||
)
|
||||
async with Client(mcp_server) as client:
|
||||
result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": "My Note: With/Invalid|Chars?",
|
||||
"folder": "my-folder",
|
||||
"content": "Testing kebab-case and invalid characters.",
|
||||
"tags": "kebab,invalid,filename",
|
||||
},
|
||||
)
|
||||
|
||||
assert len(result.content) == 1
|
||||
response_text = result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
assert len(result.content) == 1
|
||||
response_text = result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
|
||||
# File path and permalink should be kebab-case and sanitized
|
||||
assert f"project: {test_project.name}" in response_text
|
||||
assert "file_path: my-folder/my-note-with-invalid-chars.md" in response_text
|
||||
assert "permalink: my-folder/my-note-with-invalid-chars" in response_text
|
||||
assert f"[Session: Using project '{test_project.name}']" in response_text
|
||||
|
||||
# Restore original config value
|
||||
config.kebab_filenames = curr_config_val
|
||||
# File path and permalink should be kebab-case and sanitized
|
||||
assert f"project: {test_project.name}" in response_text
|
||||
assert "file_path: my-folder/my-note-with-invalid-chars.md" in response_text
|
||||
assert "permalink: my-folder/my-note-with-invalid-chars" in response_text
|
||||
assert f"[Session: Using project '{test_project.name}']" in response_text
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_write_note_kebab_filenames_repeat_invalid(mcp_server, test_project):
|
||||
async def test_write_note_kebab_filenames_repeat_invalid(mcp_server, app, test_project, app_config):
|
||||
"""Test note creation with multiple invalid and repeated characters."""
|
||||
|
||||
config = ConfigManager().config
|
||||
curr_config_val = config.kebab_filenames
|
||||
config.kebab_filenames = True
|
||||
app_config.kebab_filenames = True
|
||||
ConfigManager().save_config(app_config)
|
||||
|
||||
with patch.object(ConfigManager, "config", config):
|
||||
async with Client(mcp_server) as client:
|
||||
result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": 'Crazy<>:"|?*Note/Name',
|
||||
"folder": "my-folder",
|
||||
"content": "Should be fully kebab-case and safe.",
|
||||
"tags": "crazy,filename,test",
|
||||
},
|
||||
)
|
||||
async with Client(mcp_server) as client:
|
||||
result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": 'Crazy<>:"|?*Note/Name',
|
||||
"folder": "my-folder",
|
||||
"content": "Should be fully kebab-case and safe.",
|
||||
"tags": "crazy,filename,test",
|
||||
},
|
||||
)
|
||||
|
||||
assert len(result.content) == 1
|
||||
response_text = result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
assert len(result.content) == 1
|
||||
response_text = result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
|
||||
assert f"project: {test_project.name}" in response_text
|
||||
assert "file_path: my-folder/crazy-note-name.md" in response_text
|
||||
assert "permalink: my-folder/crazy-note-name" in response_text
|
||||
assert f"[Session: Using project '{test_project.name}']" in response_text
|
||||
|
||||
# Restore original config value
|
||||
config.kebab_filenames = curr_config_val
|
||||
assert f"project: {test_project.name}" in response_text
|
||||
assert "file_path: my-folder/crazy-note-name.md" in response_text
|
||||
assert "permalink: my-folder/crazy-note-name" in response_text
|
||||
assert f"[Session: Using project '{test_project.name}']" in response_text
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_write_note_file_path_os_path_join(mcp_server, test_project):
|
||||
async def test_write_note_file_path_os_path_join(mcp_server, app, test_project, app_config):
|
||||
"""Test that os.path.join logic in Entity.file_path works for various folder/title combinations."""
|
||||
|
||||
config = ConfigManager().config
|
||||
curr_config_val = config.kebab_filenames
|
||||
config.kebab_filenames = True
|
||||
app_config.kebab_filenames = True
|
||||
ConfigManager().save_config(app_config)
|
||||
|
||||
test_cases = [
|
||||
# (folder, title, expected file_path, expected permalink)
|
||||
@@ -407,35 +397,31 @@ async def test_write_note_file_path_os_path_join(mcp_server, test_project):
|
||||
("folder//subfolder", "Note", "folder/subfolder/note.md", "folder/subfolder/note"),
|
||||
]
|
||||
|
||||
with patch.object(ConfigManager, "config", config):
|
||||
async with Client(mcp_server) as client:
|
||||
for folder, title, expected_path, expected_permalink in test_cases:
|
||||
result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": title,
|
||||
"folder": folder,
|
||||
"content": "Testing os.path.join logic.",
|
||||
"tags": "integration,ospath",
|
||||
},
|
||||
)
|
||||
async with Client(mcp_server) as client:
|
||||
for folder, title, expected_path, expected_permalink in test_cases:
|
||||
result = await client.call_tool(
|
||||
"write_note",
|
||||
{
|
||||
"project": test_project.name,
|
||||
"title": title,
|
||||
"folder": folder,
|
||||
"content": "Testing os.path.join logic.",
|
||||
"tags": "integration,ospath",
|
||||
},
|
||||
)
|
||||
|
||||
assert len(result.content) == 1
|
||||
response_text = result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
print(response_text)
|
||||
assert len(result.content) == 1
|
||||
response_text = result.content[0].text # pyright: ignore [reportAttributeAccessIssue]
|
||||
print(response_text)
|
||||
|
||||
assert f"project: {test_project.name}" in response_text
|
||||
assert f"file_path: {expected_path}" in response_text
|
||||
assert f"permalink: {expected_permalink}" in response_text
|
||||
assert f"[Session: Using project '{test_project.name}']" in response_text
|
||||
|
||||
# Restore original config value
|
||||
config.kebab_filenames = curr_config_val
|
||||
assert f"project: {test_project.name}" in response_text
|
||||
assert f"file_path: {expected_path}" in response_text
|
||||
assert f"permalink: {expected_permalink}" in response_text
|
||||
assert f"[Session: Using project '{test_project.name}']" in response_text
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_write_note_project_path_validation(mcp_server, test_project):
|
||||
async def test_write_note_project_path_validation(mcp_server, app, test_project):
|
||||
"""Test that ProjectItem.home uses expanded path, not name (Issue #340).
|
||||
|
||||
Regression test verifying that:
|
||||
@@ -446,16 +432,12 @@ async def test_write_note_project_path_validation(mcp_server, test_project):
|
||||
the project name and path happen to be the same. The fix in src/basic_memory/schemas/project_info.py:186
|
||||
ensures .expanduser() is called, which is critical for paths with ~ like "~/Documents/Test BiSync".
|
||||
"""
|
||||
from basic_memory.schemas.project_info import ProjectItem
|
||||
from pathlib import Path
|
||||
|
||||
# Test the fix directly: ProjectItem.home should expand tilde paths
|
||||
project_with_tilde = ProjectItem(
|
||||
id=1,
|
||||
name="Test BiSync", # Name differs from path structure
|
||||
description="Test",
|
||||
path="~/Documents/Test BiSync", # Path with tilde
|
||||
is_active=True,
|
||||
is_default=False,
|
||||
)
|
||||
|
||||
|
||||
@@ -5,13 +5,15 @@ and other SQLite configuration settings work correctly in production scenarios.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import patch
|
||||
from sqlalchemy import text
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wal_mode_enabled(engine_factory):
|
||||
async def test_wal_mode_enabled(engine_factory, db_backend):
|
||||
"""Test that WAL mode is enabled on filesystem database connections."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip("SQLite-specific test - PRAGMA commands not supported in Postgres")
|
||||
|
||||
engine, _ = engine_factory
|
||||
|
||||
# Execute a query to verify WAL mode is enabled
|
||||
@@ -24,8 +26,11 @@ async def test_wal_mode_enabled(engine_factory):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_busy_timeout_configured(engine_factory):
|
||||
async def test_busy_timeout_configured(engine_factory, db_backend):
|
||||
"""Test that busy timeout is configured for database connections."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip("SQLite-specific test - PRAGMA commands not supported in Postgres")
|
||||
|
||||
engine, _ = engine_factory
|
||||
|
||||
async with engine.connect() as conn:
|
||||
@@ -37,8 +42,11 @@ async def test_busy_timeout_configured(engine_factory):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_synchronous_mode_configured(engine_factory):
|
||||
async def test_synchronous_mode_configured(engine_factory, db_backend):
|
||||
"""Test that synchronous mode is set to NORMAL for performance."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip("SQLite-specific test - PRAGMA commands not supported in Postgres")
|
||||
|
||||
engine, _ = engine_factory
|
||||
|
||||
async with engine.connect() as conn:
|
||||
@@ -50,8 +58,11 @@ async def test_synchronous_mode_configured(engine_factory):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cache_size_configured(engine_factory):
|
||||
async def test_cache_size_configured(engine_factory, db_backend):
|
||||
"""Test that cache size is configured for performance."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip("SQLite-specific test - PRAGMA commands not supported in Postgres")
|
||||
|
||||
engine, _ = engine_factory
|
||||
|
||||
async with engine.connect() as conn:
|
||||
@@ -63,8 +74,11 @@ async def test_cache_size_configured(engine_factory):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_temp_store_configured(engine_factory):
|
||||
async def test_temp_store_configured(engine_factory, db_backend):
|
||||
"""Test that temp_store is set to MEMORY."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip("SQLite-specific test - PRAGMA commands not supported in Postgres")
|
||||
|
||||
engine, _ = engine_factory
|
||||
|
||||
async with engine.connect() as conn:
|
||||
@@ -76,57 +90,63 @@ async def test_temp_store_configured(engine_factory):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_windows_locking_mode_when_on_windows(tmp_path):
|
||||
@pytest.mark.windows
|
||||
@pytest.mark.skipif(
|
||||
__import__("os").name != "nt", reason="Windows-specific test - only runs on Windows platform"
|
||||
)
|
||||
async def test_windows_locking_mode_when_on_windows(tmp_path, monkeypatch, config_manager):
|
||||
"""Test that Windows-specific locking mode is set when running on Windows."""
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from basic_memory.config import DatabaseBackend
|
||||
|
||||
# Force SQLite backend for this SQLite-specific test
|
||||
config_manager.config.database_backend = DatabaseBackend.SQLITE
|
||||
|
||||
# Set HOME environment variable
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
monkeypatch.setenv("BASIC_MEMORY_HOME", str(tmp_path / "basic-memory"))
|
||||
|
||||
db_path = tmp_path / "test_windows.db"
|
||||
|
||||
with patch("os.name", "nt"):
|
||||
# Need to patch at module level where it's imported
|
||||
with patch("basic_memory.db.os.name", "nt"):
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (
|
||||
engine,
|
||||
_,
|
||||
):
|
||||
async with engine.connect() as conn:
|
||||
result = await conn.execute(text("PRAGMA locking_mode"))
|
||||
locking_mode = result.fetchone()[0]
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (
|
||||
engine,
|
||||
_,
|
||||
):
|
||||
async with engine.connect() as conn:
|
||||
result = await conn.execute(text("PRAGMA locking_mode"))
|
||||
locking_mode = result.fetchone()[0]
|
||||
|
||||
# Locking mode should be NORMAL on Windows
|
||||
assert locking_mode.upper() == "NORMAL"
|
||||
# Locking mode should be NORMAL on Windows
|
||||
assert locking_mode.upper() == "NORMAL"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_null_pool_on_windows(tmp_path):
|
||||
@pytest.mark.windows
|
||||
@pytest.mark.skipif(
|
||||
__import__("os").name != "nt", reason="Windows-specific test - only runs on Windows platform"
|
||||
)
|
||||
async def test_null_pool_on_windows(tmp_path, monkeypatch):
|
||||
"""Test that NullPool is used on Windows to avoid connection pooling issues."""
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
# Set HOME environment variable
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
monkeypatch.setenv("BASIC_MEMORY_HOME", str(tmp_path / "basic-memory"))
|
||||
|
||||
db_path = tmp_path / "test_windows_pool.db"
|
||||
|
||||
with patch("basic_memory.db.os.name", "nt"):
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (engine, _):
|
||||
# Engine should be using NullPool on Windows
|
||||
assert isinstance(engine.pool, NullPool)
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (engine, _):
|
||||
# Engine should be using NullPool on Windows
|
||||
assert isinstance(engine.pool, NullPool)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_regular_pool_on_non_windows(tmp_path):
|
||||
"""Test that regular pooling is used on non-Windows platforms."""
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
db_path = tmp_path / "test_posix_pool.db"
|
||||
|
||||
with patch("basic_memory.db.os.name", "posix"):
|
||||
async with engine_session_factory(db_path, DatabaseType.FILESYSTEM) as (engine, _):
|
||||
# Engine should NOT be using NullPool on non-Windows
|
||||
assert not isinstance(engine.pool, NullPool)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_memory_database_no_null_pool_on_windows(tmp_path):
|
||||
@pytest.mark.windows
|
||||
@pytest.mark.skipif(
|
||||
__import__("os").name != "nt", reason="Windows-specific test - only runs on Windows platform"
|
||||
)
|
||||
async def test_memory_database_no_null_pool_on_windows(tmp_path, monkeypatch):
|
||||
"""Test that in-memory databases do NOT use NullPool even on Windows.
|
||||
|
||||
NullPool closes connections immediately, which destroys in-memory databases.
|
||||
@@ -135,9 +155,12 @@ async def test_memory_database_no_null_pool_on_windows(tmp_path):
|
||||
from basic_memory.db import engine_session_factory, DatabaseType
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
# Set HOME environment variable
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
monkeypatch.setenv("BASIC_MEMORY_HOME", str(tmp_path / "basic-memory"))
|
||||
|
||||
db_path = tmp_path / "test_memory.db"
|
||||
|
||||
with patch("basic_memory.db.os.name", "nt"):
|
||||
async with engine_session_factory(db_path, DatabaseType.MEMORY) as (engine, _):
|
||||
# In-memory databases should NOT use NullPool on Windows
|
||||
assert not isinstance(engine.pool, NullPool)
|
||||
async with engine_session_factory(db_path, DatabaseType.MEMORY) as (engine, _):
|
||||
# In-memory databases should NOT use NullPool on Windows
|
||||
assert not isinstance(engine.pool, NullPool)
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
|
||||
import pytest
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig
|
||||
from basic_memory.markdown import EntityParser, MarkdownProcessor
|
||||
from basic_memory.repository import (
|
||||
EntityRepository,
|
||||
@@ -10,7 +9,8 @@ from basic_memory.repository import (
|
||||
RelationRepository,
|
||||
ProjectRepository,
|
||||
)
|
||||
from basic_memory.repository.search_repository import SearchRepository
|
||||
from basic_memory.repository.postgres_search_repository import PostgresSearchRepository
|
||||
from basic_memory.repository.sqlite_search_repository import SQLiteSearchRepository
|
||||
from basic_memory.schemas import Entity as EntitySchema
|
||||
from basic_memory.services import FileService
|
||||
from basic_memory.services.entity_service import EntityService
|
||||
@@ -20,18 +20,25 @@ from basic_memory.sync.sync_service import SyncService
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_disable_permalinks_create_entity(tmp_path, engine_factory):
|
||||
async def test_disable_permalinks_create_entity(tmp_path, engine_factory, app_config, test_project):
|
||||
"""Test that entities created with disable_permalinks=True don't have permalinks."""
|
||||
from basic_memory.config import DatabaseBackend
|
||||
|
||||
engine, session_maker = engine_factory
|
||||
|
||||
# Create app config with disable_permalinks=True
|
||||
app_config = BasicMemoryConfig(disable_permalinks=True)
|
||||
# Override app config to enable disable_permalinks
|
||||
app_config.disable_permalinks = True
|
||||
|
||||
# Setup repositories
|
||||
entity_repository = EntityRepository(session_maker, project_id=1)
|
||||
observation_repository = ObservationRepository(session_maker, project_id=1)
|
||||
relation_repository = RelationRepository(session_maker, project_id=1)
|
||||
search_repository = SearchRepository(session_maker, project_id=1)
|
||||
entity_repository = EntityRepository(session_maker, project_id=test_project.id)
|
||||
observation_repository = ObservationRepository(session_maker, project_id=test_project.id)
|
||||
relation_repository = RelationRepository(session_maker, project_id=test_project.id)
|
||||
|
||||
# Use database-specific search repository
|
||||
if app_config.database_backend == DatabaseBackend.POSTGRES:
|
||||
search_repository = PostgresSearchRepository(session_maker, project_id=test_project.id)
|
||||
else:
|
||||
search_repository = SQLiteSearchRepository(session_maker, project_id=test_project.id)
|
||||
|
||||
# Setup services
|
||||
entity_parser = EntityParser(tmp_path)
|
||||
@@ -73,22 +80,30 @@ async def test_disable_permalinks_create_entity(tmp_path, engine_factory):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_disable_permalinks_sync_workflow(tmp_path, engine_factory):
|
||||
async def test_disable_permalinks_sync_workflow(tmp_path, engine_factory, app_config, test_project):
|
||||
"""Test full sync workflow with disable_permalinks enabled."""
|
||||
from basic_memory.config import DatabaseBackend
|
||||
|
||||
engine, session_maker = engine_factory
|
||||
|
||||
# Create app config with disable_permalinks=True
|
||||
app_config = BasicMemoryConfig(disable_permalinks=True)
|
||||
# Override app config to enable disable_permalinks
|
||||
app_config.disable_permalinks = True
|
||||
|
||||
# Create a test markdown file without frontmatter
|
||||
test_file = tmp_path / "test_note.md"
|
||||
test_file.write_text("# Test Note\nThis is test content.")
|
||||
|
||||
# Setup repositories
|
||||
entity_repository = EntityRepository(session_maker, project_id=1)
|
||||
observation_repository = ObservationRepository(session_maker, project_id=1)
|
||||
relation_repository = RelationRepository(session_maker, project_id=1)
|
||||
search_repository = SearchRepository(session_maker, project_id=1)
|
||||
entity_repository = EntityRepository(session_maker, project_id=test_project.id)
|
||||
observation_repository = ObservationRepository(session_maker, project_id=test_project.id)
|
||||
relation_repository = RelationRepository(session_maker, project_id=test_project.id)
|
||||
|
||||
# Use database-specific search repository
|
||||
if app_config.database_backend == DatabaseBackend.POSTGRES:
|
||||
search_repository = PostgresSearchRepository(session_maker, project_id=test_project.id)
|
||||
else:
|
||||
search_repository = SQLiteSearchRepository(session_maker, project_id=test_project.id)
|
||||
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
|
||||
# Setup services
|
||||
|
||||
@@ -1,369 +0,0 @@
|
||||
"""
|
||||
Performance benchmark tests for sync operations.
|
||||
|
||||
These tests measure baseline performance for indexing operations to track
|
||||
improvements from optimizations. Tests are marked with @pytest.mark.benchmark
|
||||
and can be run separately.
|
||||
|
||||
Usage:
|
||||
# Run all benchmarks
|
||||
pytest test-int/test_sync_performance_benchmark.py -v
|
||||
|
||||
# Run specific benchmark
|
||||
pytest test-int/test_sync_performance_benchmark.py::test_benchmark_sync_100_files -v
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
from pathlib import Path
|
||||
from textwrap import dedent
|
||||
|
||||
import pytest
|
||||
|
||||
from basic_memory.config import BasicMemoryConfig, ProjectConfig
|
||||
from basic_memory.sync.sync_service import get_sync_service
|
||||
|
||||
|
||||
async def create_benchmark_file(path: Path, file_num: int, total_files: int) -> None:
|
||||
"""Create a realistic test markdown file with observations and relations.
|
||||
|
||||
Args:
|
||||
path: Path to create the file at
|
||||
file_num: Current file number (for unique content)
|
||||
total_files: Total number of files being created (for relation targets)
|
||||
"""
|
||||
# Create realistic content with varying complexity
|
||||
has_relations = file_num < (total_files - 1) # Most files have relations
|
||||
num_observations = min(3 + (file_num % 5), 10) # 3-10 observations per file
|
||||
|
||||
# Generate relation targets (some will be forward references)
|
||||
relations = []
|
||||
if has_relations:
|
||||
# Reference 1-3 other files
|
||||
num_relations = min(1 + (file_num % 3), 3)
|
||||
for i in range(num_relations):
|
||||
target_num = (file_num + i + 1) % total_files
|
||||
relations.append(f"- relates_to [[test-file-{target_num:04d}]]")
|
||||
|
||||
content = dedent(f"""
|
||||
---
|
||||
type: note
|
||||
tags: [benchmark, test, category-{file_num % 10}]
|
||||
---
|
||||
# Test File {file_num:04d}
|
||||
|
||||
This is benchmark test file {file_num} of {total_files}.
|
||||
It contains realistic markdown content to simulate actual usage.
|
||||
|
||||
## Observations
|
||||
{chr(10).join([f"- [category-{i % 5}] Observation {i} for file {file_num} with some content #tag{i}" for i in range(num_observations)])}
|
||||
|
||||
## Relations
|
||||
{chr(10).join(relations) if relations else "- No relations for this file"}
|
||||
|
||||
## Additional Content
|
||||
|
||||
This section contains additional prose to simulate real documents.
|
||||
Lorem ipsum dolor sit amet, consectetur adipiscing elit. Sed do eiusmod
|
||||
tempor incididunt ut labore et dolore magna aliqua.
|
||||
|
||||
### Subsection
|
||||
|
||||
More content here to make the file realistic. This helps test the
|
||||
full indexing pipeline including content extraction and search indexing.
|
||||
""").strip()
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
|
||||
|
||||
async def generate_benchmark_files(project_dir: Path, num_files: int) -> None:
|
||||
"""Generate benchmark test files.
|
||||
|
||||
Args:
|
||||
project_dir: Directory to create files in
|
||||
num_files: Number of files to generate
|
||||
"""
|
||||
print(f"\nGenerating {num_files} test files...")
|
||||
start = time.time()
|
||||
|
||||
# Create files in batches for faster generation
|
||||
batch_size = 100
|
||||
for batch_start in range(0, num_files, batch_size):
|
||||
batch_end = min(batch_start + batch_size, num_files)
|
||||
tasks = [
|
||||
create_benchmark_file(
|
||||
project_dir / f"category-{i % 10}" / f"test-file-{i:04d}.md", i, num_files
|
||||
)
|
||||
for i in range(batch_start, batch_end)
|
||||
]
|
||||
await asyncio.gather(*tasks)
|
||||
print(f" Created files {batch_start}-{batch_end} ({batch_end}/{num_files})")
|
||||
|
||||
duration = time.time() - start
|
||||
print(f" File generation completed in {duration:.2f}s ({num_files / duration:.1f} files/sec)")
|
||||
|
||||
|
||||
def get_db_size(db_path: Path) -> tuple[int, str]:
|
||||
"""Get database file size.
|
||||
|
||||
Returns:
|
||||
Tuple of (size_bytes, formatted_size)
|
||||
"""
|
||||
if not db_path.exists():
|
||||
return 0, "0 B"
|
||||
|
||||
size_bytes = db_path.stat().st_size
|
||||
|
||||
# Format size
|
||||
for unit in ["B", "KB", "MB", "GB"]:
|
||||
if size_bytes < 1024.0:
|
||||
return size_bytes, f"{size_bytes:.2f} {unit}"
|
||||
size_bytes /= 1024.0
|
||||
|
||||
return int(size_bytes * 1024**4), f"{size_bytes:.2f} TB"
|
||||
|
||||
|
||||
async def run_sync_benchmark(
|
||||
project_config: ProjectConfig, app_config: BasicMemoryConfig, num_files: int, test_name: str
|
||||
) -> dict:
|
||||
"""Run a sync benchmark and collect metrics.
|
||||
|
||||
Args:
|
||||
project_config: Project configuration
|
||||
app_config: App configuration
|
||||
num_files: Number of files to benchmark
|
||||
test_name: Name of the test for reporting
|
||||
|
||||
Returns:
|
||||
Dictionary with benchmark results
|
||||
"""
|
||||
project_dir = project_config.home
|
||||
db_path = app_config.database_path
|
||||
|
||||
print(f"\n{'=' * 70}")
|
||||
print(f"BENCHMARK: {test_name}")
|
||||
print(f"{'=' * 70}")
|
||||
|
||||
# Generate test files
|
||||
await generate_benchmark_files(project_dir, num_files)
|
||||
|
||||
# Get initial DB size
|
||||
initial_db_size, initial_db_formatted = get_db_size(db_path)
|
||||
print(f"\nInitial database size: {initial_db_formatted}")
|
||||
|
||||
# Create sync service
|
||||
from basic_memory.repository import ProjectRepository
|
||||
from basic_memory import db
|
||||
|
||||
_, session_maker = await db.get_or_create_db(
|
||||
db_path=app_config.database_path,
|
||||
db_type=db.DatabaseType.FILESYSTEM,
|
||||
)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
|
||||
# Get or create project
|
||||
projects = await project_repository.find_all()
|
||||
if projects:
|
||||
project = projects[0]
|
||||
else:
|
||||
project = await project_repository.create(
|
||||
{
|
||||
"name": project_config.name,
|
||||
"path": str(project_config.home),
|
||||
"is_active": True,
|
||||
"is_default": True,
|
||||
}
|
||||
)
|
||||
|
||||
sync_service = await get_sync_service(project)
|
||||
|
||||
# Initialize search index (required for FTS5 table)
|
||||
await sync_service.search_service.init_search_index()
|
||||
|
||||
# Run sync and measure time
|
||||
print(f"\nStarting sync of {num_files} files...")
|
||||
sync_start = time.time()
|
||||
|
||||
report = await sync_service.sync(project_dir, project_name=project.name)
|
||||
|
||||
sync_duration = time.time() - sync_start
|
||||
|
||||
# Get final DB size
|
||||
final_db_size, final_db_formatted = get_db_size(db_path)
|
||||
db_growth = final_db_size - initial_db_size
|
||||
db_growth_formatted = f"{db_growth / 1024 / 1024:.2f} MB"
|
||||
|
||||
# Calculate metrics
|
||||
files_per_sec = num_files / sync_duration if sync_duration > 0 else 0
|
||||
ms_per_file = (sync_duration * 1000) / num_files if num_files > 0 else 0
|
||||
|
||||
# Print results
|
||||
print(f"\n{'-' * 70}")
|
||||
print("RESULTS:")
|
||||
print(f"{'-' * 70}")
|
||||
print(f"Files processed: {num_files}")
|
||||
print(f" New: {len(report.new)}")
|
||||
print(f" Modified: {len(report.modified)}")
|
||||
print(f" Deleted: {len(report.deleted)}")
|
||||
print(f" Moved: {len(report.moves)}")
|
||||
print("\nPerformance:")
|
||||
print(f" Total time: {sync_duration:.2f}s")
|
||||
print(f" Files/sec: {files_per_sec:.1f}")
|
||||
print(f" ms/file: {ms_per_file:.1f}")
|
||||
print("\nDatabase:")
|
||||
print(f" Initial size: {initial_db_formatted}")
|
||||
print(f" Final size: {final_db_formatted}")
|
||||
print(f" Growth: {db_growth_formatted}")
|
||||
print(f" Growth per file: {(db_growth / num_files / 1024):.2f} KB")
|
||||
print(f"{'=' * 70}\n")
|
||||
|
||||
return {
|
||||
"test_name": test_name,
|
||||
"num_files": num_files,
|
||||
"sync_duration_sec": sync_duration,
|
||||
"files_per_sec": files_per_sec,
|
||||
"ms_per_file": ms_per_file,
|
||||
"new_files": len(report.new),
|
||||
"modified_files": len(report.modified),
|
||||
"deleted_files": len(report.deleted),
|
||||
"moved_files": len(report.moves),
|
||||
"initial_db_size": initial_db_size,
|
||||
"final_db_size": final_db_size,
|
||||
"db_growth_bytes": db_growth,
|
||||
"db_growth_per_file_bytes": db_growth / num_files if num_files > 0 else 0,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
async def test_benchmark_sync_100_files(app_config, project_config, config_manager):
|
||||
"""Benchmark: Sync 100 files (small repository)."""
|
||||
results = await run_sync_benchmark(
|
||||
project_config, app_config, num_files=100, test_name="Sync 100 files (small repository)"
|
||||
)
|
||||
|
||||
# Basic assertions to ensure sync worked
|
||||
# Note: May be slightly more than 100 due to OS-generated files (.DS_Store, etc.)
|
||||
assert results["new_files"] >= 100
|
||||
assert results["sync_duration_sec"] > 0
|
||||
assert results["files_per_sec"] > 0
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
async def test_benchmark_sync_500_files(app_config, project_config, config_manager):
|
||||
"""Benchmark: Sync 500 files (medium repository)."""
|
||||
results = await run_sync_benchmark(
|
||||
project_config, app_config, num_files=500, test_name="Sync 500 files (medium repository)"
|
||||
)
|
||||
|
||||
# Basic assertions
|
||||
# Note: May be slightly more than 500 due to OS-generated files
|
||||
assert results["new_files"] >= 500
|
||||
assert results["sync_duration_sec"] > 0
|
||||
assert results["files_per_sec"] > 0
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.slow
|
||||
async def test_benchmark_sync_1000_files(app_config, project_config, config_manager):
|
||||
"""Benchmark: Sync 1000 files (large repository).
|
||||
|
||||
This test is marked as 'slow' and can be skipped in regular test runs:
|
||||
pytest -m "not slow"
|
||||
"""
|
||||
results = await run_sync_benchmark(
|
||||
project_config, app_config, num_files=1000, test_name="Sync 1000 files (large repository)"
|
||||
)
|
||||
|
||||
# Basic assertions
|
||||
# Note: May be slightly more than 1000 due to OS-generated files
|
||||
assert results["new_files"] >= 1000
|
||||
assert results["sync_duration_sec"] > 0
|
||||
assert results["files_per_sec"] > 0
|
||||
|
||||
|
||||
@pytest.mark.benchmark
|
||||
@pytest.mark.asyncio
|
||||
async def test_benchmark_resync_no_changes(app_config, project_config, config_manager):
|
||||
"""Benchmark: Re-sync with no changes (should be fast).
|
||||
|
||||
This tests the performance of scanning files when nothing has changed,
|
||||
which is important for cloud restarts.
|
||||
"""
|
||||
project_dir = project_config.home
|
||||
num_files = 100
|
||||
|
||||
# First sync
|
||||
print(f"\nFirst sync of {num_files} files...")
|
||||
await generate_benchmark_files(project_dir, num_files)
|
||||
|
||||
from basic_memory.repository import ProjectRepository
|
||||
from basic_memory import db
|
||||
|
||||
_, session_maker = await db.get_or_create_db(
|
||||
db_path=app_config.database_path,
|
||||
db_type=db.DatabaseType.FILESYSTEM,
|
||||
)
|
||||
project_repository = ProjectRepository(session_maker)
|
||||
projects = await project_repository.find_all()
|
||||
if projects:
|
||||
project = projects[0]
|
||||
else:
|
||||
project = await project_repository.create(
|
||||
{
|
||||
"name": project_config.name,
|
||||
"path": str(project_config.home),
|
||||
"is_active": True,
|
||||
"is_default": True,
|
||||
}
|
||||
)
|
||||
|
||||
sync_service = await get_sync_service(project)
|
||||
|
||||
# Initialize search index
|
||||
await sync_service.search_service.init_search_index()
|
||||
|
||||
await sync_service.sync(project_dir, project_name=project.name)
|
||||
|
||||
# Second sync (no changes)
|
||||
print("\nRe-sync with no changes...")
|
||||
resync_start = time.time()
|
||||
report = await sync_service.sync(project_dir, project_name=project.name)
|
||||
resync_duration = time.time() - resync_start
|
||||
|
||||
print(f"\n{'-' * 70}")
|
||||
print("RE-SYNC RESULTS (no changes):")
|
||||
print(f"{'-' * 70}")
|
||||
print(f"Files scanned: {num_files}")
|
||||
print(f"Changes detected: {report.total}")
|
||||
print(f" New: {len(report.new)}")
|
||||
print(f" Modified: {len(report.modified)}")
|
||||
print(f" Deleted: {len(report.deleted)}")
|
||||
print(f" Moved: {len(report.moves)}")
|
||||
print(f"Duration: {resync_duration:.2f}s")
|
||||
print(f"Files/sec: {num_files / resync_duration:.1f}")
|
||||
|
||||
# Debug: Show what changed
|
||||
if report.total > 0:
|
||||
print("\n⚠️ UNEXPECTED CHANGES DETECTED:")
|
||||
if report.new:
|
||||
print(f" New files ({len(report.new)}): {list(report.new)[:5]}")
|
||||
if report.modified:
|
||||
print(f" Modified files ({len(report.modified)}): {list(report.modified)[:5]}")
|
||||
if report.deleted:
|
||||
print(f" Deleted files ({len(report.deleted)}): {list(report.deleted)[:5]}")
|
||||
if report.moves:
|
||||
print(f" Moved files ({len(report.moves)}): {dict(list(report.moves.items())[:5])}")
|
||||
|
||||
print(f"{'=' * 70}\n")
|
||||
|
||||
# Should be no changes
|
||||
assert report.total == 0, (
|
||||
f"Expected no changes but got {report.total}: new={len(report.new)}, modified={len(report.modified)}, deleted={len(report.deleted)}, moves={len(report.moves)}"
|
||||
)
|
||||
assert len(report.new) == 0
|
||||
assert len(report.modified) == 0
|
||||
assert len(report.deleted) == 0
|
||||
+172
@@ -0,0 +1,172 @@
|
||||
# Dual-Backend Testing
|
||||
|
||||
Basic Memory tests run against both SQLite and Postgres backends to ensure compatibility.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```bash
|
||||
# Run tests against SQLite only (default, no setup needed)
|
||||
pytest
|
||||
|
||||
# Run tests against Postgres only (requires docker-compose)
|
||||
docker-compose -f docker-compose-postgres.yml up -d
|
||||
pytest -m postgres
|
||||
|
||||
# Run tests against BOTH backends
|
||||
docker-compose -f docker-compose-postgres.yml up -d
|
||||
pytest --run-all-backends # Not yet implemented - run both commands above
|
||||
```
|
||||
|
||||
## How It Works
|
||||
|
||||
### Parametrized Backend Fixture
|
||||
|
||||
The `db_backend` fixture is parametrized to run tests against both `sqlite` and `postgres`:
|
||||
|
||||
```python
|
||||
@pytest.fixture(
|
||||
params=[
|
||||
pytest.param("sqlite", id="sqlite"),
|
||||
pytest.param("postgres", id="postgres", marks=pytest.mark.postgres),
|
||||
]
|
||||
)
|
||||
def db_backend(request) -> Literal["sqlite", "postgres"]:
|
||||
return request.param
|
||||
```
|
||||
|
||||
### Backend-Specific Engine Factories
|
||||
|
||||
Each backend has its own engine factory implementation:
|
||||
|
||||
- **`sqlite_engine_factory`** - Uses in-memory SQLite (fast, isolated)
|
||||
- **`postgres_engine_factory`** - Uses Postgres test database (realistic, requires Docker)
|
||||
|
||||
The main `engine_factory` fixture delegates to the appropriate implementation based on `db_backend`.
|
||||
|
||||
### Configuration
|
||||
|
||||
The `app_config` fixture automatically configures the correct backend:
|
||||
|
||||
```python
|
||||
# SQLite config
|
||||
database_backend = DatabaseBackend.SQLITE
|
||||
database_url = None # Uses default SQLite path
|
||||
|
||||
# Postgres config
|
||||
database_backend = DatabaseBackend.POSTGRES
|
||||
database_url = "postgresql+asyncpg://basic_memory_user:dev_password@localhost:5433/basic_memory_test"
|
||||
```
|
||||
|
||||
## Running Postgres Tests
|
||||
|
||||
### 1. Start Postgres Docker Container
|
||||
|
||||
```bash
|
||||
docker-compose -f docker-compose-postgres.yml up -d
|
||||
```
|
||||
|
||||
This starts:
|
||||
- Postgres 17 on port **5433** (not 5432 to avoid conflicts)
|
||||
- Test database: `basic_memory_test`
|
||||
- Credentials: `basic_memory_user` / `dev_password`
|
||||
|
||||
### 2. Run Postgres Tests
|
||||
|
||||
```bash
|
||||
# Run only Postgres tests
|
||||
pytest -m postgres
|
||||
|
||||
# Run specific test with Postgres
|
||||
pytest tests/test_entity_repository.py::test_create -m postgres
|
||||
|
||||
# Skip Postgres tests (default behavior)
|
||||
pytest -m "not postgres"
|
||||
```
|
||||
|
||||
### 3. Stop Docker Container
|
||||
|
||||
```bash
|
||||
docker-compose -f docker-compose-postgres.yml down
|
||||
```
|
||||
|
||||
## Test Isolation
|
||||
|
||||
### SQLite Tests
|
||||
- Each test gets a fresh in-memory database
|
||||
- Automatic cleanup (database destroyed after test)
|
||||
- No setup required
|
||||
|
||||
### Postgres Tests
|
||||
- Database is **cleaned before each test** (drop all tables, recreate)
|
||||
- Tests share the same Postgres instance but get isolated schemas
|
||||
- Requires Docker Compose to be running
|
||||
|
||||
## Markers
|
||||
|
||||
- `postgres` - Marks tests that run against Postgres backend
|
||||
- Use `-m postgres` to run only Postgres tests
|
||||
- Use `-m "not postgres"` to skip Postgres tests (default)
|
||||
|
||||
## CI Integration
|
||||
|
||||
### GitHub Actions
|
||||
|
||||
Use service containers for Postgres (no Docker Compose needed):
|
||||
|
||||
```yaml
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
# Postgres service container
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:17
|
||||
env:
|
||||
POSTGRES_DB: basic_memory_test
|
||||
POSTGRES_USER: basic_memory_user
|
||||
POSTGRES_PASSWORD: dev_password
|
||||
ports:
|
||||
- 5433:5432
|
||||
options: >-
|
||||
--health-cmd pg_isready
|
||||
--health-interval 10s
|
||||
--health-timeout 5s
|
||||
--health-retries 5
|
||||
|
||||
steps:
|
||||
- name: Run SQLite tests
|
||||
run: pytest -m "not postgres"
|
||||
|
||||
- name: Run Postgres tests
|
||||
run: pytest -m postgres
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Postgres tests fail with "connection refused"
|
||||
|
||||
Make sure Docker Compose is running:
|
||||
```bash
|
||||
docker-compose -f docker-compose-postgres.yml ps
|
||||
docker-compose -f docker-compose-postgres.yml logs postgres
|
||||
```
|
||||
|
||||
### Port 5433 already in use
|
||||
|
||||
Either:
|
||||
- Stop the conflicting service
|
||||
- Change the port in `docker-compose-postgres.yml` and `tests/conftest.py`
|
||||
|
||||
### Tests hang or timeout
|
||||
|
||||
Check Postgres health:
|
||||
```bash
|
||||
docker-compose -f docker-compose-postgres.yml exec postgres pg_isready -U basic_memory_user
|
||||
```
|
||||
|
||||
## Future Enhancements
|
||||
|
||||
- [ ] Add `--run-all-backends` CLI flag to run both backends in sequence
|
||||
- [ ] Implement test fixtures for backend-specific features (e.g., Postgres full-text search vs SQLite FTS5)
|
||||
- [ ] Add performance comparison benchmarks between backends
|
||||
@@ -9,17 +9,19 @@ from basic_memory.mcp.async_client import create_client
|
||||
|
||||
|
||||
def test_create_client_uses_asgi_when_no_remote_env():
|
||||
"""Test that create_client uses ASGI transport when BASIC_MEMORY_USE_REMOTE_API is not set."""
|
||||
# Ensure env vars are not set (pop if they exist)
|
||||
"""Test that create_client uses ASGI transport when cloud mode is disabled."""
|
||||
# Ensure env vars are not set and config cloud_mode is False
|
||||
with patch.dict("os.environ", clear=False):
|
||||
os.environ.pop("BASIC_MEMORY_USE_REMOTE_API", None)
|
||||
os.environ.pop("BASIC_MEMORY_CLOUD_MODE", None)
|
||||
|
||||
client = create_client()
|
||||
# Also patch the config's cloud_mode to ensure it's False
|
||||
with patch.object(ConfigManager().config, "cloud_mode", False):
|
||||
client = create_client()
|
||||
|
||||
assert isinstance(client, AsyncClient)
|
||||
assert isinstance(client._transport, ASGITransport)
|
||||
assert str(client.base_url) == "http://test"
|
||||
assert isinstance(client, AsyncClient)
|
||||
assert isinstance(client._transport, ASGITransport)
|
||||
assert str(client.base_url) == "http://test"
|
||||
|
||||
|
||||
def test_create_client_uses_http_when_cloud_mode_env_set():
|
||||
@@ -37,16 +39,18 @@ def test_create_client_uses_http_when_cloud_mode_env_set():
|
||||
|
||||
def test_create_client_configures_extended_timeouts():
|
||||
"""Test that create_client configures 30-second timeouts for long operations."""
|
||||
# Ensure env vars are not set (pop if they exist)
|
||||
# Ensure env vars are not set and config cloud_mode is False
|
||||
with patch.dict("os.environ", clear=False):
|
||||
os.environ.pop("BASIC_MEMORY_USE_REMOTE_API", None)
|
||||
os.environ.pop("BASIC_MEMORY_CLOUD_MODE", None)
|
||||
|
||||
client = create_client()
|
||||
# Also patch the config's cloud_mode to ensure it's False
|
||||
with patch.object(ConfigManager().config, "cloud_mode", False):
|
||||
client = create_client()
|
||||
|
||||
# Verify timeout configuration
|
||||
assert isinstance(client.timeout, Timeout)
|
||||
assert client.timeout.connect == 10.0 # 10 seconds for connection
|
||||
assert client.timeout.read == 30.0 # 30 seconds for reading
|
||||
assert client.timeout.write == 30.0 # 30 seconds for writing
|
||||
assert client.timeout.pool == 30.0 # 30 seconds for pool
|
||||
# Verify timeout configuration
|
||||
assert isinstance(client.timeout, Timeout)
|
||||
assert client.timeout.connect == 10.0 # 10 seconds for connection
|
||||
assert client.timeout.read == 30.0 # 30 seconds for reading
|
||||
assert client.timeout.write == 30.0 # 30 seconds for writing
|
||||
assert client.timeout.pool == 30.0 # 30 seconds for pool
|
||||
|
||||
@@ -18,6 +18,7 @@ def template_loader():
|
||||
def entity_summary():
|
||||
"""Create a sample EntitySummary for testing."""
|
||||
return EntitySummary(
|
||||
entity_id=1,
|
||||
title="Test Entity",
|
||||
permalink="test/entity",
|
||||
type=SearchItemType.ENTITY,
|
||||
@@ -34,6 +35,8 @@ def context_with_results(entity_summary):
|
||||
|
||||
# Create an observation for the entity
|
||||
observation = ObservationSummary(
|
||||
observation_id=1,
|
||||
entity_id=1,
|
||||
title="Test Observation",
|
||||
permalink="test/entity/observations/1",
|
||||
category="test",
|
||||
|
||||
@@ -219,20 +219,20 @@ async def test_update_project_path_endpoint(test_config, client, project_service
|
||||
test_project_name = "test-update-project"
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
test_root = Path(temp_dir)
|
||||
old_path = (test_root / "old-location").as_posix()
|
||||
new_path = (test_root / "new-location").as_posix()
|
||||
old_path = test_root / "old-location"
|
||||
new_path = test_root / "new-location"
|
||||
|
||||
await project_service.add_project(test_project_name, old_path)
|
||||
await project_service.add_project(test_project_name, str(old_path))
|
||||
|
||||
try:
|
||||
# Verify initial state
|
||||
project = await project_service.get_project(test_project_name)
|
||||
assert project is not None
|
||||
assert project.path == old_path
|
||||
assert Path(project.path) == old_path
|
||||
|
||||
# Update the project path
|
||||
response = await client.patch(
|
||||
f"{project_url}/project/{test_project_name}", json={"path": new_path}
|
||||
f"{project_url}/project/{test_project_name}", json={"path": str(new_path)}
|
||||
)
|
||||
|
||||
# Verify response
|
||||
@@ -248,16 +248,16 @@ async def test_update_project_path_endpoint(test_config, client, project_service
|
||||
|
||||
# Check old project data
|
||||
assert data["old_project"]["name"] == test_project_name
|
||||
assert data["old_project"]["path"] == old_path
|
||||
assert Path(data["old_project"]["path"]) == old_path
|
||||
|
||||
# Check new project data
|
||||
assert data["new_project"]["name"] == test_project_name
|
||||
assert data["new_project"]["path"] == new_path
|
||||
assert Path(data["new_project"]["path"]) == new_path
|
||||
|
||||
# Verify project was actually updated in database
|
||||
updated_project = await project_service.get_project(test_project_name)
|
||||
assert updated_project is not None
|
||||
assert updated_project.path == new_path
|
||||
assert Path(updated_project.path) == new_path
|
||||
|
||||
finally:
|
||||
# Clean up
|
||||
|
||||
@@ -218,9 +218,13 @@ async def test_get_resource_entities(client, project_config, entity_repository,
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_resource_entities_pagination(
|
||||
client, project_config, entity_repository, project_url
|
||||
client, project_config, entity_repository, project_url, db_backend
|
||||
):
|
||||
"""Test getting content by permalink match."""
|
||||
if db_backend == "postgres":
|
||||
pytest.skip(
|
||||
"Pagination differs: relations expand to multiple entities, ordering is undefined"
|
||||
)
|
||||
# Create entity
|
||||
content1 = "# Test Content\n"
|
||||
data = {
|
||||
|
||||
@@ -12,7 +12,7 @@ from basic_memory.schemas.search import SearchItemType, SearchResponse
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def indexed_entity(init_search_index, full_entity, search_service):
|
||||
async def indexed_entity(full_entity, search_service):
|
||||
"""Create an entity and index it."""
|
||||
await search_service.index_entity(full_entity)
|
||||
return full_entity
|
||||
@@ -118,8 +118,16 @@ async def test_search_empty(search_service, client, project_url):
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reindex(client, search_service, entity_service, session_maker, project_url):
|
||||
async def test_reindex(
|
||||
client, search_service, entity_service, session_maker, project_url, app_config
|
||||
):
|
||||
"""Test reindex endpoint."""
|
||||
# Skip for Postgres - needs investigation of database connection isolation
|
||||
from basic_memory.config import DatabaseBackend
|
||||
|
||||
if app_config.database_backend == DatabaseBackend.POSTGRES:
|
||||
pytest.skip("Not yet supported for Postgres - database connection isolation issue")
|
||||
|
||||
# Create test entity and document
|
||||
await entity_service.create_entity(
|
||||
EntitySchema(
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
"""V2 API tests."""
|
||||
@@ -0,0 +1,21 @@
|
||||
"""Fixtures for V2 API tests."""
|
||||
|
||||
import pytest
|
||||
|
||||
from basic_memory.models import Project
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def v2_project_url(test_project: Project) -> str:
|
||||
"""Create a URL prefix for v2 project-scoped routes using project ID.
|
||||
|
||||
This helps tests generate the correct URL for v2 project-scoped routes
|
||||
which use integer project IDs instead of permalinks.
|
||||
"""
|
||||
return f"/v2/projects/{test_project.id}"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def v2_projects_url() -> str:
|
||||
"""Base URL for v2 project management endpoints."""
|
||||
return "/v2/projects"
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Tests for V2 directory API routes (ID-based endpoints)."""
|
||||
|
||||
import pytest
|
||||
from httpx import AsyncClient
|
||||
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.schemas.directory import DirectoryNode
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_directory_tree(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test getting directory tree via v2 endpoint."""
|
||||
response = await client.get(f"{v2_project_url}/directory/tree")
|
||||
|
||||
assert response.status_code == 200
|
||||
tree = DirectoryNode.model_validate(response.json())
|
||||
assert tree.type == "directory"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_directory_structure(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test getting directory structure (folders only) via v2 endpoint."""
|
||||
response = await client.get(f"{v2_project_url}/directory/structure")
|
||||
|
||||
assert response.status_code == 200
|
||||
structure = DirectoryNode.model_validate(response.json())
|
||||
assert structure.type == "directory"
|
||||
# Structure should only contain directories, not files
|
||||
if structure.children:
|
||||
for child in structure.children:
|
||||
assert child.type == "directory"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_directory_default(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test listing directory contents with default parameters via v2 endpoint."""
|
||||
response = await client.get(f"{v2_project_url}/directory/list")
|
||||
|
||||
assert response.status_code == 200
|
||||
nodes = response.json()
|
||||
assert isinstance(nodes, list)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_directory_with_depth(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test listing directory with custom depth via v2 endpoint."""
|
||||
response = await client.get(f"{v2_project_url}/directory/list?depth=2")
|
||||
|
||||
assert response.status_code == 200
|
||||
nodes = response.json()
|
||||
assert isinstance(nodes, list)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_directory_with_glob(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test listing directory with file name glob filter via v2 endpoint."""
|
||||
response = await client.get(f"{v2_project_url}/directory/list?file_name_glob=*.md")
|
||||
|
||||
assert response.status_code == 200
|
||||
nodes = response.json()
|
||||
assert isinstance(nodes, list)
|
||||
# All file nodes should have .md extension
|
||||
for node in nodes:
|
||||
if node.get("type") == "file":
|
||||
assert node.get("path", "").endswith(".md")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_directory_with_custom_path(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test listing a specific directory path via v2 endpoint."""
|
||||
response = await client.get(f"{v2_project_url}/directory/list?dir_name=/")
|
||||
|
||||
assert response.status_code == 200
|
||||
nodes = response.json()
|
||||
assert isinstance(nodes, list)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_directory_invalid_project_id(
|
||||
client: AsyncClient,
|
||||
):
|
||||
"""Test directory endpoints with invalid project ID return 404."""
|
||||
# Test tree endpoint
|
||||
response = await client.get("/v2/projects/999999/directory/tree")
|
||||
assert response.status_code == 404
|
||||
|
||||
# Test structure endpoint
|
||||
response = await client.get("/v2/projects/999999/directory/structure")
|
||||
assert response.status_code == 404
|
||||
|
||||
# Test list endpoint
|
||||
response = await client.get("/v2/projects/999999/directory/list")
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_v2_directory_endpoints_use_project_id_not_name(
|
||||
client: AsyncClient, test_project: Project
|
||||
):
|
||||
"""Verify v2 directory endpoints require project ID, not name."""
|
||||
# Try using project name instead of ID - should fail
|
||||
response = await client.get(f"/v2/projects/{test_project.name}/directory/tree")
|
||||
|
||||
# Should get validation error or 404 because name is not a valid integer
|
||||
assert response.status_code in [404, 422]
|
||||
@@ -0,0 +1,530 @@
|
||||
"""Tests for V2 importer API routes (ID-based endpoints)."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from httpx import AsyncClient
|
||||
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.schemas.importer import (
|
||||
ChatImportResult,
|
||||
EntityImportResult,
|
||||
ProjectImportResult,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def chatgpt_json_content():
|
||||
"""Sample ChatGPT conversation data for testing."""
|
||||
return [
|
||||
{
|
||||
"title": "Test Conversation",
|
||||
"create_time": 1736616594.24054,
|
||||
"update_time": 1736616603.164995,
|
||||
"mapping": {
|
||||
"root": {"id": "root", "message": None, "parent": None, "children": ["msg1"]},
|
||||
"msg1": {
|
||||
"id": "msg1",
|
||||
"message": {
|
||||
"id": "msg1",
|
||||
"author": {"role": "user", "name": None, "metadata": {}},
|
||||
"create_time": 1736616594.24054,
|
||||
"content": {
|
||||
"content_type": "text",
|
||||
"parts": ["Hello, this is a test message"],
|
||||
},
|
||||
"status": "finished_successfully",
|
||||
"metadata": {},
|
||||
},
|
||||
"parent": "root",
|
||||
"children": ["msg2"],
|
||||
},
|
||||
"msg2": {
|
||||
"id": "msg2",
|
||||
"message": {
|
||||
"id": "msg2",
|
||||
"author": {"role": "assistant", "name": None, "metadata": {}},
|
||||
"create_time": 1736616603.164995,
|
||||
"content": {"content_type": "text", "parts": ["This is a test response"]},
|
||||
"status": "finished_successfully",
|
||||
"metadata": {},
|
||||
},
|
||||
"parent": "msg1",
|
||||
"children": [],
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def claude_conversations_json_content():
|
||||
"""Sample Claude conversations data for testing."""
|
||||
return [
|
||||
{
|
||||
"uuid": "test-uuid",
|
||||
"name": "Test Conversation",
|
||||
"created_at": "2025-01-05T20:55:32.499880+00:00",
|
||||
"updated_at": "2025-01-05T20:56:39.477600+00:00",
|
||||
"chat_messages": [
|
||||
{
|
||||
"uuid": "msg-1",
|
||||
"text": "Hello, this is a test",
|
||||
"sender": "human",
|
||||
"created_at": "2025-01-05T20:55:32.499880+00:00",
|
||||
"content": [{"type": "text", "text": "Hello, this is a test"}],
|
||||
},
|
||||
{
|
||||
"uuid": "msg-2",
|
||||
"text": "Response to test",
|
||||
"sender": "assistant",
|
||||
"created_at": "2025-01-05T20:55:40.123456+00:00",
|
||||
"content": [{"type": "text", "text": "Response to test"}],
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def claude_projects_json_content():
|
||||
"""Sample Claude projects data for testing."""
|
||||
return [
|
||||
{
|
||||
"uuid": "test-uuid",
|
||||
"name": "Test Project",
|
||||
"created_at": "2025-01-05T20:55:32.499880+00:00",
|
||||
"updated_at": "2025-01-05T20:56:39.477600+00:00",
|
||||
"prompt_template": "# Test Prompt\n\nThis is a test prompt.",
|
||||
"docs": [
|
||||
{
|
||||
"uuid": "doc-uuid-1",
|
||||
"filename": "Test Document",
|
||||
"content": "# Test Document\n\nThis is test content.",
|
||||
"created_at": "2025-01-05T20:56:39.477600+00:00",
|
||||
},
|
||||
{
|
||||
"uuid": "doc-uuid-2",
|
||||
"filename": "Another Document",
|
||||
"content": "# Another Document\n\nMore test content.",
|
||||
"created_at": "2025-01-05T20:56:39.477600+00:00",
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def memory_json_content():
|
||||
"""Sample memory.json data for testing."""
|
||||
return [
|
||||
{
|
||||
"type": "entity",
|
||||
"name": "test_entity",
|
||||
"entityType": "test",
|
||||
"observations": ["Test observation 1", "Test observation 2"],
|
||||
},
|
||||
{
|
||||
"type": "relation",
|
||||
"from": "test_entity",
|
||||
"to": "related_entity",
|
||||
"relationType": "test_relation",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
async def create_test_upload_file(tmp_path, content):
|
||||
"""Create a test file for upload."""
|
||||
file_path = tmp_path / "test_import.json"
|
||||
with open(file_path, "w", encoding="utf-8") as f:
|
||||
json.dump(content, f)
|
||||
|
||||
return file_path
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_chatgpt(
|
||||
project_config,
|
||||
client: AsyncClient,
|
||||
tmp_path,
|
||||
chatgpt_json_content,
|
||||
file_service,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test importing ChatGPT conversations via v2 endpoint."""
|
||||
# Create a test file
|
||||
file_path = await create_test_upload_file(tmp_path, chatgpt_json_content)
|
||||
|
||||
# Create a multipart form with the file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("conversations.json", f, "application/json")}
|
||||
data = {"folder": "test_chatgpt"}
|
||||
|
||||
# Send request
|
||||
response = await client.post(f"{v2_project_url}/import/chatgpt", files=files, data=data)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 200
|
||||
result = ChatImportResult.model_validate(response.json())
|
||||
assert result.success is True
|
||||
assert result.conversations == 1
|
||||
assert result.messages == 2
|
||||
|
||||
# Verify files were created
|
||||
conv_path = Path("test_chatgpt") / "20250111-Test_Conversation.md"
|
||||
assert await file_service.exists(conv_path)
|
||||
|
||||
content, _ = await file_service.read_file(conv_path)
|
||||
assert "# Test Conversation" in content
|
||||
assert "Hello, this is a test message" in content
|
||||
assert "This is a test response" in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_chatgpt_invalid_file(client: AsyncClient, tmp_path, v2_project_url: str):
|
||||
"""Test importing invalid ChatGPT file via v2 endpoint."""
|
||||
# Create invalid file
|
||||
file_path = tmp_path / "invalid.json"
|
||||
with open(file_path, "w") as f:
|
||||
f.write("This is not JSON")
|
||||
|
||||
# Create multipart form with invalid file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("invalid.json", f, "application/json")}
|
||||
data = {"folder": "test_chatgpt"}
|
||||
|
||||
# Send request - this should return an error
|
||||
response = await client.post(f"{v2_project_url}/import/chatgpt", files=files, data=data)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 500
|
||||
assert "Import failed" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_claude_conversations(
|
||||
client: AsyncClient,
|
||||
tmp_path,
|
||||
claude_conversations_json_content,
|
||||
file_service,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test importing Claude conversations via v2 endpoint."""
|
||||
# Create a test file
|
||||
file_path = await create_test_upload_file(tmp_path, claude_conversations_json_content)
|
||||
|
||||
# Create a multipart form with the file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("conversations.json", f, "application/json")}
|
||||
data = {"folder": "test_claude_conversations"}
|
||||
|
||||
# Send request
|
||||
response = await client.post(
|
||||
f"{v2_project_url}/import/claude/conversations", files=files, data=data
|
||||
)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 200
|
||||
result = ChatImportResult.model_validate(response.json())
|
||||
assert result.success is True
|
||||
assert result.conversations == 1
|
||||
assert result.messages == 2
|
||||
|
||||
# Verify files were created
|
||||
conv_path = Path("test_claude_conversations") / "20250105-Test_Conversation.md"
|
||||
assert await file_service.exists(conv_path)
|
||||
|
||||
content, _ = await file_service.read_file(conv_path)
|
||||
assert "# Test Conversation" in content
|
||||
assert "Hello, this is a test" in content
|
||||
assert "Response to test" in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_claude_conversations_invalid_file(
|
||||
client: AsyncClient, tmp_path, v2_project_url: str
|
||||
):
|
||||
"""Test importing invalid Claude conversations file via v2 endpoint."""
|
||||
# Create invalid file
|
||||
file_path = tmp_path / "invalid.json"
|
||||
with open(file_path, "w") as f:
|
||||
f.write("This is not JSON")
|
||||
|
||||
# Create multipart form with invalid file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("invalid.json", f, "application/json")}
|
||||
data = {"folder": "test_claude_conversations"}
|
||||
|
||||
# Send request - this should return an error
|
||||
response = await client.post(
|
||||
f"{v2_project_url}/import/claude/conversations", files=files, data=data
|
||||
)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 500
|
||||
assert "Import failed" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_claude_projects(
|
||||
client: AsyncClient, tmp_path, claude_projects_json_content, file_service, v2_project_url: str
|
||||
):
|
||||
"""Test importing Claude projects via v2 endpoint."""
|
||||
# Create a test file
|
||||
file_path = await create_test_upload_file(tmp_path, claude_projects_json_content)
|
||||
|
||||
# Create a multipart form with the file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("projects.json", f, "application/json")}
|
||||
data = {"folder": "test_claude_projects"}
|
||||
|
||||
# Send request
|
||||
response = await client.post(
|
||||
f"{v2_project_url}/import/claude/projects", files=files, data=data
|
||||
)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 200
|
||||
result = ProjectImportResult.model_validate(response.json())
|
||||
assert result.success is True
|
||||
assert result.documents == 2
|
||||
assert result.prompts == 1
|
||||
|
||||
# Verify files were created
|
||||
project_dir = Path("test_claude_projects") / "Test_Project"
|
||||
assert await file_service.exists(project_dir / "prompt-template.md")
|
||||
assert await file_service.exists(project_dir / "docs" / "Test_Document.md")
|
||||
assert await file_service.exists(project_dir / "docs" / "Another_Document.md")
|
||||
|
||||
# Check content
|
||||
prompt_content, _ = await file_service.read_file(project_dir / "prompt-template.md")
|
||||
assert "# Test Prompt" in prompt_content
|
||||
|
||||
doc_content, _ = await file_service.read_file(project_dir / "docs" / "Test_Document.md")
|
||||
assert "# Test Document" in doc_content
|
||||
assert "This is test content" in doc_content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_claude_projects_invalid_file(
|
||||
client: AsyncClient, tmp_path, v2_project_url: str
|
||||
):
|
||||
"""Test importing invalid Claude projects file via v2 endpoint."""
|
||||
# Create invalid file
|
||||
file_path = tmp_path / "invalid.json"
|
||||
with open(file_path, "w") as f:
|
||||
f.write("This is not JSON")
|
||||
|
||||
# Create multipart form with invalid file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("invalid.json", f, "application/json")}
|
||||
data = {"folder": "test_claude_projects"}
|
||||
|
||||
# Send request - this should return an error
|
||||
response = await client.post(
|
||||
f"{v2_project_url}/import/claude/projects", files=files, data=data
|
||||
)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 500
|
||||
assert "Import failed" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_memory_json(
|
||||
client: AsyncClient, tmp_path, memory_json_content, file_service, v2_project_url: str
|
||||
):
|
||||
"""Test importing memory.json file via v2 endpoint."""
|
||||
# Create a test file
|
||||
json_file = tmp_path / "memory.json"
|
||||
with open(json_file, "w", encoding="utf-8") as f:
|
||||
for entity in memory_json_content:
|
||||
f.write(json.dumps(entity) + "\n")
|
||||
|
||||
# Create a multipart form with the file
|
||||
with open(json_file, "rb") as f:
|
||||
files = {"file": ("memory.json", f, "application/json")}
|
||||
data = {"folder": "test_memory_json"}
|
||||
|
||||
# Send request
|
||||
response = await client.post(f"{v2_project_url}/import/memory-json", files=files, data=data)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 200
|
||||
result = EntityImportResult.model_validate(response.json())
|
||||
assert result.success is True
|
||||
assert result.entities == 1
|
||||
assert result.relations == 1
|
||||
|
||||
# Verify files were created
|
||||
entity_path = Path("test_memory_json") / "test" / "test_entity.md"
|
||||
assert await file_service.exists(entity_path)
|
||||
|
||||
# Check content
|
||||
content, _ = await file_service.read_file(entity_path)
|
||||
assert "Test observation 1" in content
|
||||
assert "Test observation 2" in content
|
||||
assert "test_relation [[related_entity]]" in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_memory_json_without_folder(
|
||||
client: AsyncClient, tmp_path, memory_json_content, file_service, v2_project_url: str
|
||||
):
|
||||
"""Test importing memory.json file without specifying a destination folder."""
|
||||
# Create a test file
|
||||
json_file = tmp_path / "memory.json"
|
||||
with open(json_file, "w", encoding="utf-8") as f:
|
||||
for entity in memory_json_content:
|
||||
f.write(json.dumps(entity) + "\n")
|
||||
|
||||
# Create a multipart form with the file
|
||||
with open(json_file, "rb") as f:
|
||||
files = {"file": ("memory.json", f, "application/json")}
|
||||
|
||||
# Send request without destination_folder
|
||||
response = await client.post(f"{v2_project_url}/import/memory-json", files=files)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 200
|
||||
result = EntityImportResult.model_validate(response.json())
|
||||
assert result.success is True
|
||||
assert result.entities == 1
|
||||
assert result.relations == 1
|
||||
|
||||
# Verify files were created in the default directory
|
||||
entity_path = Path("conversations") / "test" / "test_entity.md"
|
||||
assert await file_service.exists(entity_path)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_memory_json_invalid_file(client: AsyncClient, tmp_path, v2_project_url: str):
|
||||
"""Test importing invalid memory.json file via v2 endpoint."""
|
||||
# Create invalid file
|
||||
file_path = tmp_path / "invalid.json"
|
||||
with open(file_path, "w") as f:
|
||||
f.write("This is not JSON")
|
||||
|
||||
# Create multipart form with invalid file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("invalid.json", f, "application/json")}
|
||||
data = {"folder": "test_memory_json"}
|
||||
|
||||
# Send request - this should return an error
|
||||
response = await client.post(f"{v2_project_url}/import/memory-json", files=files, data=data)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 500
|
||||
assert "Import failed" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_v2_import_endpoints_use_project_id_not_name(
|
||||
client: AsyncClient, tmp_path, test_project: Project, chatgpt_json_content
|
||||
):
|
||||
"""Verify v2 import endpoints require project ID, not name."""
|
||||
# Create a test file
|
||||
file_path = await create_test_upload_file(tmp_path, chatgpt_json_content)
|
||||
|
||||
# Try using project name instead of ID - should fail
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("conversations.json", f, "application/json")}
|
||||
data = {"folder": "test"}
|
||||
|
||||
response = await client.post(
|
||||
f"/v2/projects/{test_project.name}/import/chatgpt",
|
||||
files=files,
|
||||
data=data,
|
||||
)
|
||||
|
||||
# Should get validation error or 404 because name is not a valid integer
|
||||
assert response.status_code in [404, 422]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_invalid_project_id(client: AsyncClient, tmp_path, chatgpt_json_content):
|
||||
"""Test import endpoints with invalid project ID return 404."""
|
||||
# Create a test file
|
||||
file_path = await create_test_upload_file(tmp_path, chatgpt_json_content)
|
||||
|
||||
# Test all import endpoints
|
||||
endpoints = [
|
||||
"/import/chatgpt",
|
||||
"/import/claude/conversations",
|
||||
"/import/claude/projects",
|
||||
"/import/memory-json",
|
||||
]
|
||||
|
||||
for endpoint in endpoints:
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("test.json", f, "application/json")}
|
||||
data = {"folder": "test"}
|
||||
|
||||
response = await client.post(
|
||||
f"/v2/projects/999999{endpoint}",
|
||||
files=files,
|
||||
data=data,
|
||||
)
|
||||
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_missing_file(client: AsyncClient, v2_project_url: str):
|
||||
"""Test importing with missing file via v2 endpoint."""
|
||||
# Send a request without a file
|
||||
response = await client.post(f"{v2_project_url}/import/chatgpt", data={"folder": "test_folder"})
|
||||
|
||||
# Check that the request was rejected
|
||||
assert response.status_code in [400, 422] # Either bad request or unprocessable entity
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_empty_file(client: AsyncClient, tmp_path, v2_project_url: str):
|
||||
"""Test importing an empty file via v2 endpoint."""
|
||||
# Create an empty file
|
||||
file_path = tmp_path / "empty.json"
|
||||
with open(file_path, "w") as f:
|
||||
f.write("")
|
||||
|
||||
# Create multipart form with empty file
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("empty.json", f, "application/json")}
|
||||
data = {"folder": "test_chatgpt"}
|
||||
|
||||
# Send request
|
||||
response = await client.post(f"{v2_project_url}/import/chatgpt", files=files, data=data)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 500
|
||||
assert "Import failed" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_import_malformed_json(client: AsyncClient, tmp_path, v2_project_url: str):
|
||||
"""Test importing malformed JSON for all v2 import endpoints."""
|
||||
# Create malformed JSON file
|
||||
file_path = tmp_path / "malformed.json"
|
||||
with open(file_path, "w") as f:
|
||||
f.write('{"incomplete": "json"') # Missing closing brace
|
||||
|
||||
# Test all import endpoints
|
||||
endpoints = [
|
||||
(f"{v2_project_url}/import/chatgpt", {"folder": "test"}),
|
||||
(f"{v2_project_url}/import/claude/conversations", {"folder": "test"}),
|
||||
(f"{v2_project_url}/import/claude/projects", {"folder": "test"}),
|
||||
(f"{v2_project_url}/import/memory-json", {"folder": "test"}),
|
||||
]
|
||||
|
||||
for endpoint, data in endpoints:
|
||||
# Create multipart form with malformed JSON
|
||||
with open(file_path, "rb") as f:
|
||||
files = {"file": ("malformed.json", f, "application/json")}
|
||||
|
||||
# Send request
|
||||
response = await client.post(endpoint, files=files, data=data)
|
||||
|
||||
# Check response
|
||||
assert response.status_code == 500
|
||||
assert "Import failed" in response.json()["detail"]
|
||||
@@ -0,0 +1,407 @@
|
||||
"""Tests for V2 knowledge graph API routes (ID-based endpoints)."""
|
||||
|
||||
import pytest
|
||||
from httpx import AsyncClient
|
||||
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.schemas import DeleteEntitiesResponse
|
||||
from basic_memory.schemas.v2 import EntityResponseV2, EntityResolveResponse
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_identifier_by_permalink(
|
||||
client: AsyncClient, test_graph, v2_project_url, test_project: Project, entity_repository
|
||||
):
|
||||
"""Test resolving an identifier by permalink returns correct entity ID."""
|
||||
# test_graph fixture creates some test entities
|
||||
# We'll use one of them to test resolution
|
||||
|
||||
# Create an entity first
|
||||
entity_data = {
|
||||
"title": "TestResolve",
|
||||
"folder": "test",
|
||||
"content": "Test content for resolve",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=entity_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
entity_id = created_entity.id
|
||||
|
||||
# Now resolve it by permalink
|
||||
resolve_data = {"identifier": created_entity.permalink}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 200
|
||||
resolved = EntityResolveResponse.model_validate(response.json())
|
||||
assert resolved.entity_id == entity_id
|
||||
assert resolved.permalink == created_entity.permalink
|
||||
assert resolved.resolution_method == "permalink"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_identifier_not_found(client: AsyncClient, v2_project_url):
|
||||
"""Test resolving a non-existent identifier returns 404."""
|
||||
resolve_data = {"identifier": "nonexistent/entity"}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/resolve", json=resolve_data)
|
||||
|
||||
assert response.status_code == 404
|
||||
assert "Could not resolve identifier" in response.json()["detail"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_entity_by_id(client: AsyncClient, test_graph, v2_project_url, entity_repository):
|
||||
"""Test getting an entity by its numeric ID."""
|
||||
# Create an entity first
|
||||
entity_data = {
|
||||
"title": "TestGetById",
|
||||
"folder": "test",
|
||||
"content": "Test content for get by ID",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=entity_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
entity_id = created_entity.id
|
||||
|
||||
# Get it by ID using v2 endpoint
|
||||
response = await client.get(f"{v2_project_url}/knowledge/entities/{entity_id}")
|
||||
|
||||
assert response.status_code == 200
|
||||
entity = EntityResponseV2.model_validate(response.json())
|
||||
assert entity.id == entity_id
|
||||
assert entity.title == "TestGetById"
|
||||
assert entity.api_version == "v2"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_entity_by_id_not_found(client: AsyncClient, v2_project_url):
|
||||
"""Test getting a non-existent entity by ID returns 404."""
|
||||
response = await client.get(f"{v2_project_url}/knowledge/entities/999999")
|
||||
|
||||
assert response.status_code == 404
|
||||
assert "not found" in response.json()["detail"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_entity(client: AsyncClient, file_service, v2_project_url):
|
||||
"""Test creating an entity via v2 endpoint."""
|
||||
data = {
|
||||
"title": "TestV2Entity",
|
||||
"folder": "test",
|
||||
"entity_type": "test",
|
||||
"content_type": "text/markdown",
|
||||
"content": "TestContent for V2",
|
||||
}
|
||||
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=data)
|
||||
|
||||
assert response.status_code == 200
|
||||
entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 endpoints must return id field
|
||||
assert entity.id is not None
|
||||
assert isinstance(entity.id, int)
|
||||
assert entity.api_version == "v2"
|
||||
|
||||
assert entity.permalink == "test/test-v2-entity"
|
||||
assert entity.file_path == "test/TestV2Entity.md"
|
||||
assert entity.entity_type == data["entity_type"]
|
||||
|
||||
# Verify file was created
|
||||
file_path = file_service.get_entity_path(entity)
|
||||
file_content, _ = await file_service.read_file(file_path)
|
||||
assert data["content"] in file_content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_entity_with_observations_and_relations(
|
||||
client: AsyncClient, file_service, v2_project_url
|
||||
):
|
||||
"""Test creating an entity with observations and relations via v2."""
|
||||
data = {
|
||||
"title": "TestV2Complex",
|
||||
"folder": "test",
|
||||
"content": """
|
||||
# TestV2Complex
|
||||
|
||||
## Observations
|
||||
- [note] This is a test observation #tag1 (context)
|
||||
- related to [[OtherEntity]]
|
||||
""",
|
||||
}
|
||||
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=data)
|
||||
|
||||
assert response.status_code == 200
|
||||
entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 endpoints must return id field
|
||||
assert entity.id is not None
|
||||
assert isinstance(entity.id, int)
|
||||
assert entity.api_version == "v2"
|
||||
|
||||
assert len(entity.observations) == 1
|
||||
assert entity.observations[0].category == "note"
|
||||
assert entity.observations[0].content == "This is a test observation #tag1"
|
||||
assert entity.observations[0].tags == ["tag1"]
|
||||
|
||||
assert len(entity.relations) == 1
|
||||
assert entity.relations[0].relation_type == "related to"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_entity_by_id(
|
||||
client: AsyncClient, file_service, v2_project_url, entity_repository
|
||||
):
|
||||
"""Test updating an entity by ID using PUT (replace)."""
|
||||
# Create an entity first
|
||||
create_data = {
|
||||
"title": "TestUpdate",
|
||||
"folder": "test",
|
||||
"content": "Original content",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=create_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
original_id = created_entity.id
|
||||
|
||||
# Update it by ID
|
||||
update_data = {
|
||||
"title": "TestUpdate",
|
||||
"folder": "test",
|
||||
"content": "Updated content via V2",
|
||||
}
|
||||
response = await client.put(
|
||||
f"{v2_project_url}/knowledge/entities/{original_id}",
|
||||
json=update_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
updated_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 update must return id field
|
||||
assert updated_entity.id is not None
|
||||
assert isinstance(updated_entity.id, int)
|
||||
assert updated_entity.api_version == "v2"
|
||||
|
||||
# Verify file was updated
|
||||
file_path = file_service.get_entity_path(updated_entity)
|
||||
file_content, _ = await file_service.read_file(file_path)
|
||||
assert "Updated content via V2" in file_content
|
||||
assert "Original content" not in file_content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_edit_entity_by_id_append(
|
||||
client: AsyncClient, file_service, v2_project_url, entity_repository
|
||||
):
|
||||
"""Test editing an entity by ID using PATCH (append operation)."""
|
||||
# Create an entity first
|
||||
create_data = {
|
||||
"title": "TestEdit",
|
||||
"folder": "test",
|
||||
"content": "# TestEdit\n\nOriginal content",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=create_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
original_id = created_entity.id
|
||||
|
||||
# Edit it by appending
|
||||
edit_data = {
|
||||
"operation": "append",
|
||||
"content": "\n\n## New Section\n\nAppended content",
|
||||
}
|
||||
response = await client.patch(
|
||||
f"{v2_project_url}/knowledge/entities/{original_id}",
|
||||
json=edit_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
edited_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 patch must return id field
|
||||
assert edited_entity.id is not None
|
||||
assert isinstance(edited_entity.id, int)
|
||||
assert edited_entity.api_version == "v2"
|
||||
|
||||
# Verify file has both original and appended content
|
||||
file_path = file_service.get_entity_path(edited_entity)
|
||||
file_content, _ = await file_service.read_file(file_path)
|
||||
assert "Original content" in file_content
|
||||
assert "Appended content" in file_content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_edit_entity_by_id_find_replace(
|
||||
client: AsyncClient, file_service, v2_project_url, entity_repository
|
||||
):
|
||||
"""Test editing an entity by ID using PATCH (find/replace operation)."""
|
||||
# Create an entity first
|
||||
create_data = {
|
||||
"title": "TestFindReplace",
|
||||
"folder": "test",
|
||||
"content": "# TestFindReplace\n\nOld text that will be replaced",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=create_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
original_id = created_entity.id
|
||||
|
||||
# Edit using find/replace
|
||||
edit_data = {
|
||||
"operation": "find_replace",
|
||||
"find_text": "Old text",
|
||||
"content": "New text",
|
||||
}
|
||||
response = await client.patch(
|
||||
f"{v2_project_url}/knowledge/entities/{original_id}",
|
||||
json=edit_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
edited_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 patch must return id field
|
||||
assert edited_entity.id is not None
|
||||
assert isinstance(edited_entity.id, int)
|
||||
assert edited_entity.api_version == "v2"
|
||||
|
||||
# Verify replacement
|
||||
file_path = file_service.get_entity_path(created_entity)
|
||||
file_content, _ = await file_service.read_file(file_path)
|
||||
assert "New text" in file_content
|
||||
assert "Old text" not in file_content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_entity_by_id(
|
||||
client: AsyncClient, file_service, v2_project_url, entity_repository
|
||||
):
|
||||
"""Test deleting an entity by ID."""
|
||||
# Create an entity first
|
||||
create_data = {
|
||||
"title": "TestDelete",
|
||||
"folder": "test",
|
||||
"content": "Content to be deleted",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=create_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
entity_id = created_entity.id
|
||||
|
||||
# Delete it by ID
|
||||
response = await client.delete(f"{v2_project_url}/knowledge/entities/{entity_id}")
|
||||
|
||||
assert response.status_code == 200
|
||||
delete_response = DeleteEntitiesResponse.model_validate(response.json())
|
||||
assert delete_response.deleted is True
|
||||
|
||||
# Verify it's gone - trying to get it should return 404
|
||||
response = await client.get(f"{v2_project_url}/knowledge/entities/{entity_id}")
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_entity_by_id_not_found(client: AsyncClient, v2_project_url):
|
||||
"""Test deleting a non-existent entity returns deleted=False (idempotent)."""
|
||||
response = await client.delete(f"{v2_project_url}/knowledge/entities/999999")
|
||||
|
||||
# Delete is idempotent - returns 200 with deleted=False
|
||||
assert response.status_code == 200
|
||||
delete_response = DeleteEntitiesResponse.model_validate(response.json())
|
||||
assert delete_response.deleted is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_move_entity(client: AsyncClient, file_service, v2_project_url, entity_repository):
|
||||
"""Test moving an entity to a new location."""
|
||||
# Create an entity first
|
||||
create_data = {
|
||||
"title": "TestMove",
|
||||
"folder": "test",
|
||||
"content": "Content to be moved",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=create_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id
|
||||
assert created_entity.id is not None
|
||||
original_id = created_entity.id
|
||||
|
||||
# Move it to a new folder (V2 uses entity ID in path)
|
||||
move_data = {
|
||||
"destination_path": "moved/MovedEntity.md",
|
||||
}
|
||||
response = await client.put(
|
||||
f"{v2_project_url}/knowledge/entities/{created_entity.id}/move", json=move_data
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
moved_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 move must return id field
|
||||
assert moved_entity.id is not None
|
||||
assert isinstance(moved_entity.id, int)
|
||||
assert moved_entity.api_version == "v2"
|
||||
|
||||
# ID should remain the same (stable reference)
|
||||
assert moved_entity.id == original_id
|
||||
assert moved_entity.file_path == "moved/MovedEntity.md"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_v2_endpoints_use_project_id_not_name(client: AsyncClient, test_project: Project):
|
||||
"""Verify v2 endpoints require project ID, not name."""
|
||||
# Try using project name instead of ID - should fail
|
||||
response = await client.get(f"/v2/{test_project.name}/knowledge/entities/1")
|
||||
|
||||
# Should get validation error or 404 because name is not a valid integer
|
||||
assert response.status_code in [404, 422]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_entity_response_v2_has_api_version(
|
||||
client: AsyncClient, v2_project_url, entity_repository
|
||||
):
|
||||
"""Test that EntityResponseV2 includes api_version field."""
|
||||
# Create an entity
|
||||
entity_data = {
|
||||
"title": "TestApiVersion",
|
||||
"folder": "test",
|
||||
"content": "Test content",
|
||||
}
|
||||
response = await client.post(f"{v2_project_url}/knowledge/entities", json=entity_data)
|
||||
assert response.status_code == 200
|
||||
created_entity = EntityResponseV2.model_validate(response.json())
|
||||
|
||||
# V2 create must return id and api_version
|
||||
assert created_entity.id is not None
|
||||
assert created_entity.api_version == "v2"
|
||||
entity_id = created_entity.id
|
||||
|
||||
# Get it via v2 endpoint
|
||||
response = await client.get(f"{v2_project_url}/knowledge/entities/{entity_id}")
|
||||
assert response.status_code == 200
|
||||
|
||||
entity_v2 = EntityResponseV2.model_validate(response.json())
|
||||
assert entity_v2.api_version == "v2"
|
||||
assert entity_v2.id == entity_id
|
||||
@@ -0,0 +1,301 @@
|
||||
"""Tests for v2 memory router endpoints."""
|
||||
|
||||
import pytest
|
||||
from httpx import AsyncClient
|
||||
from pathlib import Path
|
||||
|
||||
from basic_memory.models import Project
|
||||
|
||||
|
||||
async def create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
):
|
||||
"""Helper to create an entity with file and index it."""
|
||||
# Create file
|
||||
test_content = f"# {entity_data['title']}\n\nTest content"
|
||||
file_path = Path(test_project.path) / entity_data["file_path"]
|
||||
file_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
await file_service.write_file(file_path, test_content)
|
||||
|
||||
# Create entity
|
||||
entity = await entity_repository.create(entity_data)
|
||||
|
||||
# Index for search
|
||||
await search_service.index_entity(entity)
|
||||
|
||||
return entity
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_recent_context(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test getting recent activity context."""
|
||||
entity_data = {
|
||||
"title": "Recent Test Entity",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "recent_test.md",
|
||||
"checksum": "abc123",
|
||||
}
|
||||
await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get recent context
|
||||
response = await client.get(f"{v2_project_url}/memory/recent")
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
|
||||
# Verify response structure (GraphContext uses 'results' not 'entities')
|
||||
assert "results" in data
|
||||
assert "metadata" in data
|
||||
assert "page" in data
|
||||
assert "page_size" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_recent_context_with_pagination(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test recent context with pagination parameters."""
|
||||
# Create multiple test entities
|
||||
for i in range(5):
|
||||
entity_data = {
|
||||
"title": f"Entity {i}",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": f"entity_{i}.md",
|
||||
"checksum": f"checksum{i}",
|
||||
}
|
||||
await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get recent context with pagination
|
||||
response = await client.get(
|
||||
f"{v2_project_url}/memory/recent", params={"page": 1, "page_size": 3}
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
assert data["page"] == 1
|
||||
assert data["page_size"] == 3
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_recent_context_with_type_filter(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test filtering recent context by type."""
|
||||
# Create a test entity
|
||||
entity_data = {
|
||||
"title": "Filtered Entity",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "filtered.md",
|
||||
"checksum": "xyz789",
|
||||
}
|
||||
await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get recent context filtered by type
|
||||
response = await client.get(f"{v2_project_url}/memory/recent", params={"type": ["entity"]})
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_recent_context_with_timeframe(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test recent context with custom timeframe."""
|
||||
response = await client.get(f"{v2_project_url}/memory/recent", params={"timeframe": "1d"})
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_recent_context_invalid_project_id(
|
||||
client: AsyncClient,
|
||||
):
|
||||
"""Test getting recent context with invalid project ID returns 404."""
|
||||
response = await client.get("/v2/projects/999999/memory/recent")
|
||||
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_memory_context_by_permalink(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test getting context for a specific memory URI (permalink)."""
|
||||
# Create a test entity
|
||||
entity_data = {
|
||||
"title": "Context Test",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "context_test.md",
|
||||
"checksum": "def456",
|
||||
"permalink": "context-test",
|
||||
}
|
||||
await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get context for this entity
|
||||
response = await client.get(f"{v2_project_url}/memory/context-test")
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_memory_context_by_id(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test getting context using ID-based memory URI."""
|
||||
# Create a test entity
|
||||
entity_data = {
|
||||
"title": "ID Context Test",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "id_context_test.md",
|
||||
"checksum": "ghi789",
|
||||
}
|
||||
created_entity = await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get context using ID format (memory://id/123 or memory://123)
|
||||
response = await client.get(f"{v2_project_url}/memory/id/{created_entity.id}")
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_memory_context_with_depth(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test getting context with depth parameter."""
|
||||
# Create a test entity
|
||||
entity_data = {
|
||||
"title": "Depth Test",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "depth_test.md",
|
||||
"checksum": "jkl012",
|
||||
"permalink": "depth-test",
|
||||
}
|
||||
await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get context with depth
|
||||
response = await client.get(f"{v2_project_url}/memory/depth-test", params={"depth": 2})
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_memory_context_not_found(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
):
|
||||
"""Test getting context for non-existent memory URI returns 404."""
|
||||
response = await client.get(f"{v2_project_url}/memory/nonexistent-uri")
|
||||
|
||||
# Note: This might return 200 with empty results depending on implementation
|
||||
# Adjust assertion based on actual behavior
|
||||
assert response.status_code in [200, 404]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_memory_context_with_timeframe(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
v2_project_url: str,
|
||||
entity_repository,
|
||||
search_service,
|
||||
file_service,
|
||||
):
|
||||
"""Test getting context with timeframe filter."""
|
||||
# Create a test entity
|
||||
entity_data = {
|
||||
"title": "Timeframe Test",
|
||||
"entity_type": "note",
|
||||
"content_type": "text/markdown",
|
||||
"file_path": "timeframe_test.md",
|
||||
"checksum": "mno345",
|
||||
"permalink": "timeframe-test",
|
||||
}
|
||||
await create_test_entity(
|
||||
test_project, entity_data, entity_repository, search_service, file_service
|
||||
)
|
||||
|
||||
# Get context with timeframe
|
||||
response = await client.get(
|
||||
f"{v2_project_url}/memory/timeframe-test", params={"timeframe": "7d"}
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "results" in data
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_v2_memory_endpoints_use_project_id_not_name(
|
||||
client: AsyncClient,
|
||||
test_project: Project,
|
||||
):
|
||||
"""Test that v2 memory endpoints reject string project names."""
|
||||
# Try to use project name instead of ID - should fail
|
||||
response = await client.get(f"/v2/{test_project.name}/memory/recent")
|
||||
|
||||
# FastAPI path validation should reject non-integer project_id
|
||||
assert response.status_code in [404, 422]
|
||||
@@ -0,0 +1,251 @@
|
||||
"""Tests for V2 project management API routes (ID-based endpoints)."""
|
||||
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from httpx import AsyncClient
|
||||
|
||||
from basic_memory.models import Project
|
||||
from basic_memory.schemas.project_info import ProjectItem, ProjectStatusResponse
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_project_by_id(client: AsyncClient, test_project: Project, v2_projects_url):
|
||||
"""Test getting a project by its numeric ID."""
|
||||
response = await client.get(f"{v2_projects_url}/{test_project.id}")
|
||||
|
||||
assert response.status_code == 200
|
||||
project = ProjectItem.model_validate(response.json())
|
||||
assert project.id == test_project.id
|
||||
assert project.name == test_project.name
|
||||
assert project.path == test_project.path
|
||||
assert project.is_default == (test_project.is_default or False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_project_by_id_not_found(client: AsyncClient, v2_projects_url):
|
||||
"""Test getting a non-existent project by ID returns 404."""
|
||||
response = await client.get(f"{v2_projects_url}/999999")
|
||||
|
||||
assert response.status_code == 404
|
||||
assert "not found" in response.json()["detail"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_path_by_id(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url
|
||||
):
|
||||
"""Test updating a project's path by ID."""
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
new_path = str(Path(tmpdir) / "new-project-location")
|
||||
Path(new_path).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
update_data = {"path": new_path}
|
||||
response = await client.patch(
|
||||
f"{v2_projects_url}/{test_project.id}",
|
||||
json=update_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
status_response = ProjectStatusResponse.model_validate(response.json())
|
||||
assert status_response.status == "success"
|
||||
assert status_response.new_project.id == test_project.id
|
||||
# Normalize paths for cross-platform comparison (Windows uses backslashes, API returns forward slashes)
|
||||
assert Path(status_response.new_project.path) == Path(new_path)
|
||||
assert status_response.old_project.id == test_project.id
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_invalid_path(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url
|
||||
):
|
||||
"""Test updating with a relative path returns 400."""
|
||||
update_data = {"path": "relative/path"}
|
||||
response = await client.patch(
|
||||
f"{v2_projects_url}/{test_project.id}",
|
||||
json=update_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 400
|
||||
assert "absolute" in response.json()["detail"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_not_found(client: AsyncClient, v2_projects_url):
|
||||
"""Test updating a non-existent project returns 404."""
|
||||
update_data = {"path": "/tmp/new-path"}
|
||||
response = await client.patch(
|
||||
f"{v2_projects_url}/999999",
|
||||
json=update_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_set_default_project_by_id(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url, project_repository, project_service
|
||||
):
|
||||
"""Test setting a project as default by ID."""
|
||||
# Create a second project to test setting default
|
||||
await project_service.add_project("second-project", "/tmp/second-project")
|
||||
|
||||
# Get the created project from the repository to get its ID
|
||||
created_project = await project_repository.get_by_name("second-project")
|
||||
assert created_project is not None
|
||||
|
||||
# Set the second project as default
|
||||
response = await client.put(f"{v2_projects_url}/{created_project.id}/default")
|
||||
|
||||
assert response.status_code == 200
|
||||
status_response = ProjectStatusResponse.model_validate(response.json())
|
||||
assert status_response.status == "success"
|
||||
assert status_response.default is True
|
||||
assert status_response.new_project.id == created_project.id
|
||||
assert status_response.new_project.is_default is True
|
||||
assert status_response.old_project.id == test_project.id
|
||||
assert status_response.old_project.is_default is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_set_default_project_not_found(client: AsyncClient, v2_projects_url):
|
||||
"""Test setting a non-existent project as default returns 404."""
|
||||
response = await client.put(f"{v2_projects_url}/999999/default")
|
||||
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_project_by_id(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url, project_repository, project_service
|
||||
):
|
||||
"""Test deleting a project by ID."""
|
||||
# Create a second project since we can't delete the default
|
||||
await project_service.add_project("to-delete", "/tmp/to-delete")
|
||||
|
||||
# Get the created project from the repository to get its ID
|
||||
created_project = await project_repository.get_by_name("to-delete")
|
||||
assert created_project is not None
|
||||
|
||||
# Delete it
|
||||
response = await client.delete(f"{v2_projects_url}/{created_project.id}")
|
||||
|
||||
assert response.status_code == 200
|
||||
status_response = ProjectStatusResponse.model_validate(response.json())
|
||||
assert status_response.status == "success"
|
||||
assert status_response.old_project.id == created_project.id
|
||||
assert status_response.new_project is None
|
||||
|
||||
# Verify it's deleted - trying to get it should return 404
|
||||
response = await client.get(f"{v2_projects_url}/{created_project.id}")
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_project_with_delete_notes_param(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url, project_repository, project_service
|
||||
):
|
||||
"""Test deleting a project with delete_notes parameter."""
|
||||
# Create a project in a temp directory
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
project_path = Path(tmpdir) / "test-delete-notes"
|
||||
project_path.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Create a test file in the project
|
||||
test_file = project_path / "test.md"
|
||||
test_file.write_text("Test content")
|
||||
|
||||
await project_service.add_project("delete-with-notes", str(project_path))
|
||||
|
||||
# Get the created project from the repository to get its ID
|
||||
created_project = await project_repository.get_by_name("delete-with-notes")
|
||||
assert created_project is not None
|
||||
|
||||
# Delete with delete_notes=true
|
||||
response = await client.delete(f"{v2_projects_url}/{created_project.id}?delete_notes=true")
|
||||
|
||||
assert response.status_code == 200
|
||||
|
||||
# Verify directory was deleted
|
||||
assert not project_path.exists()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_default_project_fails(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url
|
||||
):
|
||||
"""Test that deleting the default project returns 400."""
|
||||
# test_project is the default project
|
||||
response = await client.delete(f"{v2_projects_url}/{test_project.id}")
|
||||
|
||||
assert response.status_code == 400
|
||||
assert "default project" in response.json()["detail"].lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_delete_project_not_found(client: AsyncClient, v2_projects_url):
|
||||
"""Test deleting a non-existent project returns 404."""
|
||||
response = await client.delete(f"{v2_projects_url}/999999")
|
||||
|
||||
assert response.status_code == 404
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_v2_project_endpoints_use_id_not_name(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url
|
||||
):
|
||||
"""Verify v2 project endpoints require project ID, not name."""
|
||||
# Try using project name instead of ID - should fail
|
||||
response = await client.get(f"{v2_projects_url}/{test_project.name}")
|
||||
|
||||
# Should get 404 or 422 because name is not a valid integer
|
||||
assert response.status_code in [404, 422]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_project_id_stability_after_rename(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url, project_repository
|
||||
):
|
||||
"""Test that project ID remains stable even after renaming."""
|
||||
original_id = test_project.id
|
||||
original_name = test_project.name
|
||||
|
||||
# Get project by ID
|
||||
response = await client.get(f"{v2_projects_url}/{original_id}")
|
||||
assert response.status_code == 200
|
||||
project_before = ProjectItem.model_validate(response.json())
|
||||
assert project_before.id == original_id
|
||||
assert project_before.name == original_name
|
||||
|
||||
# Even if we renamed the project (not testing rename here, just the concept),
|
||||
# the ID would stay the same. This test demonstrates the stability.
|
||||
# Re-fetch by same ID
|
||||
response = await client.get(f"{v2_projects_url}/{original_id}")
|
||||
assert response.status_code == 200
|
||||
project_after = ProjectItem.model_validate(response.json())
|
||||
assert project_after.id == original_id
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_project_active_status(
|
||||
client: AsyncClient, test_project: Project, v2_projects_url, project_repository, project_service
|
||||
):
|
||||
"""Test updating a project's active status by ID."""
|
||||
# Create a non-default project
|
||||
await project_service.add_project("test-active", "/tmp/test-active")
|
||||
|
||||
# Get the created project from the repository to get its ID
|
||||
created_project = await project_repository.get_by_name("test-active")
|
||||
assert created_project is not None
|
||||
|
||||
# Update active status
|
||||
update_data = {"is_active": False}
|
||||
response = await client.patch(
|
||||
f"{v2_projects_url}/{created_project.id}",
|
||||
json=update_data,
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
status_response = ProjectStatusResponse.model_validate(response.json())
|
||||
assert status_response.status == "success"
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user